A high-speed inference platform optimized for low per-user latency on frontier open-weight models.
6 of 6
# pin every request to Makora
curl https://www.ninjachat.ai/api/v1/chat/completions \
-H "Authorization: Bearer $NINJACHAT_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "qwen-3.8-flash-next",
"messages": [{ "role": "user", "content": "Hello!" }],
"routing": { "providers": { "only": ["makora"] } }
}'