curl --request POST \
--url https://api.v2.healthproximate.com/api/v1/llm/chat/completions \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "<string>",
"messages": [
{
"content": "<string>",
"tool_calls": [
{}
],
"tool_call_id": "<string>"
}
],
"max_tokens": 1024,
"temperature": 123,
"top_p": 123,
"stop": "<string>",
"stream": false,
"reasoning": {},
"redact_only": false,
"tools": [
{}
],
"tool_choice": "<unknown>",
"conversation_key": "<string>"
}
'import requests
url = "https://api.v2.healthproximate.com/api/v1/llm/chat/completions"
payload = {
"model": "<string>",
"messages": [
{
"content": "<string>",
"tool_calls": [{}],
"tool_call_id": "<string>"
}
],
"max_tokens": 1024,
"temperature": 123,
"top_p": 123,
"stop": "<string>",
"stream": False,
"reasoning": {},
"redact_only": False,
"tools": [{}],
"tool_choice": "<unknown>",
"conversation_key": "<string>"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: '<string>',
messages: [{content: '<string>', tool_calls: [{}], tool_call_id: '<string>'}],
max_tokens: 1024,
temperature: 123,
top_p: 123,
stop: '<string>',
stream: false,
reasoning: {},
redact_only: false,
tools: [{}],
tool_choice: '<unknown>',
conversation_key: '<string>'
})
};
fetch('https://api.v2.healthproximate.com/api/v1/llm/chat/completions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.v2.healthproximate.com/api/v1/llm/chat/completions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => '<string>',
'messages' => [
[
'content' => '<string>',
'tool_calls' => [
[
]
],
'tool_call_id' => '<string>'
]
],
'max_tokens' => 1024,
'temperature' => 123,
'top_p' => 123,
'stop' => '<string>',
'stream' => false,
'reasoning' => [
],
'redact_only' => false,
'tools' => [
[
]
],
'tool_choice' => '<unknown>',
'conversation_key' => '<string>'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.v2.healthproximate.com/api/v1/llm/chat/completions"
payload := strings.NewReader("{\n \"model\": \"<string>\",\n \"messages\": [\n {\n \"content\": \"<string>\",\n \"tool_calls\": [\n {}\n ],\n \"tool_call_id\": \"<string>\"\n }\n ],\n \"max_tokens\": 1024,\n \"temperature\": 123,\n \"top_p\": 123,\n \"stop\": \"<string>\",\n \"stream\": false,\n \"reasoning\": {},\n \"redact_only\": false,\n \"tools\": [\n {}\n ],\n \"tool_choice\": \"<unknown>\",\n \"conversation_key\": \"<string>\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.v2.healthproximate.com/api/v1/llm/chat/completions")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"<string>\",\n \"messages\": [\n {\n \"content\": \"<string>\",\n \"tool_calls\": [\n {}\n ],\n \"tool_call_id\": \"<string>\"\n }\n ],\n \"max_tokens\": 1024,\n \"temperature\": 123,\n \"top_p\": 123,\n \"stop\": \"<string>\",\n \"stream\": false,\n \"reasoning\": {},\n \"redact_only\": false,\n \"tools\": [\n {}\n ],\n \"tool_choice\": \"<unknown>\",\n \"conversation_key\": \"<string>\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.v2.healthproximate.com/api/v1/llm/chat/completions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"<string>\",\n \"messages\": [\n {\n \"content\": \"<string>\",\n \"tool_calls\": [\n {}\n ],\n \"tool_call_id\": \"<string>\"\n }\n ],\n \"max_tokens\": 1024,\n \"temperature\": 123,\n \"top_p\": 123,\n \"stop\": \"<string>\",\n \"stream\": false,\n \"reasoning\": {},\n \"redact_only\": false,\n \"tools\": [\n {}\n ],\n \"tool_choice\": \"<unknown>\",\n \"conversation_key\": \"<string>\"\n}"
response = http.request(request)
puts response.read_bodyPHI-safe chat completion (OpenAI-compatible)
An OpenAI-compatible chat completion endpoint that redacts PHI before the model sees it and restores the real values in the answer.
client = OpenAI(base_url="https://<host>/api/v1/llm", api_key="hpx_...")
client.chat.completions.create(model="<model>", messages=[...])
Per request: your message content is redacted (Malaysian IC/MRN/passport/phone, names, emails, account numbers — clinical content such as ages, dates like “day 3 post-op”, vitals and dosages is preserved), the placeholders go to the model, and the model’s answer has your values substituted back. The mapping lives only in the request’s memory and is never stored.
Every response carries redaction_receipt, including rehydration_complete
and unresolved_placeholder_count — if the model altered our placeholders, the
receipt says so rather than returning residue silently.
Tool calling is supported: send OpenAI-style tools, receive tool_calls
with your values re-hydrated. Your tool role results are redacted on the way
out like any other message, and so are the arguments of an assistant turn you
send back.
Long answers: stream them. A non-streamed request must complete inside a
fixed budget, and a model that has not answered by then returns 504 with
error.type: "model_timeout" naming the model. The GPT-5.x models are the ones
this reaches: their latency on an identical prompt varies widely, measured
between 1.4s and 106.2s. A streamed request is not subject to that budget at
all, because output arrives continuously — so for long generations send
stream=true together with redact_only=true.
max_tokens minimums. The GPT-5.x models refuse anything below 16 and
the request is rejected with 400 before it reaches the provider. Other model
families have no such floor.
Not supported (refused, not silently dropped): streaming while re-hydrating
(stream with redact_only=false), tools combined with stream, non-text
content parts, forcing a specific tool via tool_choice, and the deprecated
functions/function_call form.
curl --request POST \
--url https://api.v2.healthproximate.com/api/v1/llm/chat/completions \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "<string>",
"messages": [
{
"content": "<string>",
"tool_calls": [
{}
],
"tool_call_id": "<string>"
}
],
"max_tokens": 1024,
"temperature": 123,
"top_p": 123,
"stop": "<string>",
"stream": false,
"reasoning": {},
"redact_only": false,
"tools": [
{}
],
"tool_choice": "<unknown>",
"conversation_key": "<string>"
}
'import requests
url = "https://api.v2.healthproximate.com/api/v1/llm/chat/completions"
payload = {
"model": "<string>",
"messages": [
{
"content": "<string>",
"tool_calls": [{}],
"tool_call_id": "<string>"
}
],
"max_tokens": 1024,
"temperature": 123,
"top_p": 123,
"stop": "<string>",
"stream": False,
"reasoning": {},
"redact_only": False,
"tools": [{}],
"tool_choice": "<unknown>",
"conversation_key": "<string>"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: '<string>',
messages: [{content: '<string>', tool_calls: [{}], tool_call_id: '<string>'}],
max_tokens: 1024,
temperature: 123,
top_p: 123,
stop: '<string>',
stream: false,
reasoning: {},
redact_only: false,
tools: [{}],
tool_choice: '<unknown>',
conversation_key: '<string>'
})
};
fetch('https://api.v2.healthproximate.com/api/v1/llm/chat/completions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.v2.healthproximate.com/api/v1/llm/chat/completions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => '<string>',
'messages' => [
[
'content' => '<string>',
'tool_calls' => [
[
]
],
'tool_call_id' => '<string>'
]
],
'max_tokens' => 1024,
'temperature' => 123,
'top_p' => 123,
'stop' => '<string>',
'stream' => false,
'reasoning' => [
],
'redact_only' => false,
'tools' => [
[
]
],
'tool_choice' => '<unknown>',
'conversation_key' => '<string>'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.v2.healthproximate.com/api/v1/llm/chat/completions"
payload := strings.NewReader("{\n \"model\": \"<string>\",\n \"messages\": [\n {\n \"content\": \"<string>\",\n \"tool_calls\": [\n {}\n ],\n \"tool_call_id\": \"<string>\"\n }\n ],\n \"max_tokens\": 1024,\n \"temperature\": 123,\n \"top_p\": 123,\n \"stop\": \"<string>\",\n \"stream\": false,\n \"reasoning\": {},\n \"redact_only\": false,\n \"tools\": [\n {}\n ],\n \"tool_choice\": \"<unknown>\",\n \"conversation_key\": \"<string>\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.v2.healthproximate.com/api/v1/llm/chat/completions")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"<string>\",\n \"messages\": [\n {\n \"content\": \"<string>\",\n \"tool_calls\": [\n {}\n ],\n \"tool_call_id\": \"<string>\"\n }\n ],\n \"max_tokens\": 1024,\n \"temperature\": 123,\n \"top_p\": 123,\n \"stop\": \"<string>\",\n \"stream\": false,\n \"reasoning\": {},\n \"redact_only\": false,\n \"tools\": [\n {}\n ],\n \"tool_choice\": \"<unknown>\",\n \"conversation_key\": \"<string>\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.v2.healthproximate.com/api/v1/llm/chat/completions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"<string>\",\n \"messages\": [\n {\n \"content\": \"<string>\",\n \"tool_calls\": [\n {}\n ],\n \"tool_call_id\": \"<string>\"\n }\n ],\n \"max_tokens\": 1024,\n \"temperature\": 123,\n \"top_p\": 123,\n \"stop\": \"<string>\",\n \"stream\": false,\n \"reasoning\": {},\n \"redact_only\": false,\n \"tools\": [\n {}\n ],\n \"tool_choice\": \"<unknown>\",\n \"conversation_key\": \"<string>\"\n}"
response = http.request(request)
puts response.read_bodyAuthorizations
Your hpx_ API key as a Bearer token, so OpenAI-compatible SDKs work unmodified. X-API-Key and first-party JWTs are also accepted.
Body
A model id from the registry. External keys may only use models approved for PHI traffic.
1Show child attributes
Show child attributes
1 <= x <= 320001Reasoning-model controls, e.g. {"effort": "high", "summary": "detailed"}. Only effort and summary are forwarded; other keys are ignored. Applies to the GPT-5.x family; ignored by other models.
Show child attributes
Show child attributes
Return the model's answer WITHOUT substituting your values back (placeholders remain). Required if you want streaming, and useful when the answer is going somewhere that should not hold PHI.
OpenAI-style function tools. Forwarded to the model; returned tool_calls come back with your values re-hydrated.
Show child attributes
Show child attributes
"auto" (default) or "none". Forcing a specific call ("required", or a named function) is refused rather than downgraded to "auto", so a forced call never appears to have been honoured when it was not.
Supply the same value across turns so a given person keeps the same placeholder; multi-turn clients resend history and renaming mid-conversation confuses the model.
200Response
chat.completion with a redaction receipt attached

