Skip to content

Python

All examples use requests (pip install requests) and read the key from the environment. They are runnable as-is against the live API.

import os
import time
import requests
BASE = "https://ghostmind.optdmsa.com"
KEY = os.environ["GHOSTMIND_API_KEY"]
AUTH = {"Authorization": f"Bearer {KEY}"}

Chat Completions

response = requests.post(
f"{BASE}/v1/chat/completions",
headers={**AUTH, "Content-Type": "application/json"},
json={
"model": "gpt-4o",
"messages": [{"role": "user", "content": "Hello!"}],
},
timeout=180, # upstream latency is typically 5–15 s; see Performance notes
)
print(response.json()["choices"][0]["message"]["content"])
# The conversation id for follow-up messages:
conv_id = response.headers["x-ghostmind-conversation-id"]

Streaming

with requests.post(
f"{BASE}/v1/chat/completions",
headers={**AUTH, "Content-Type": "application/json"},
json={
"model": "gpt-4o",
"stream": True,
"messages": [{"role": "user", "content": "Count to five."}],
},
stream=True,
timeout=180,
) as response:
for line in response.iter_lines():
if line:
print(line.decode()) # SSE data: lines, ending with data: [DONE]

Durable conversation

Capture the conversation id once, continue anytime — same upstream thread.

first = requests.post(
f"{BASE}/v1/chat/completions",
headers={**AUTH, "Content-Type": "application/json"},
json={"model": "gpt-4o", "messages": [{"role": "user",
"content": "Remember: the invoice prefix is RT-."}]},
timeout=180,
)
conv_id = first.headers["x-ghostmind-conversation-id"]
later = requests.post(
f"{BASE}/v1/chat/completions",
headers={**AUTH,
"X-GhostMind-Conversation-Id": conv_id,
"Content-Type": "application/json"},
json={"model": "gpt-4o", "messages": [{"role": "user",
"content": "What is the invoice prefix?"}]},
timeout=180,
)

Project-scoped chat with a source file

The upstream-project slug (not the UUID) goes in X-GhostMind-Upstream-Project.

proj = requests.post(
f"{BASE}/v1/projects",
headers={**AUTH, "Content-Type": "application/json"},
json={"name": "My Engine",
"instructions": "Always answer in JSON.",
"memory_policy": "project_only"},
).json()
slug, proj_id = proj["slug"], proj["id"]
with open("brand_voice.txt", "rb") as f:
file = requests.post(
f"{BASE}/v1/projects/{proj_id}/files",
headers=AUTH, files={"file": f},
).json()
# Wait until the file is indexed ("ready") — required for retrieval.
while True:
detail = requests.get(f"{BASE}/v1/projects/{proj_id}", headers=AUTH).json()
status = next(s["status"] for s in detail["sources"] if s["id"] == file["id"])
if status == "ready":
break
if status == "failed":
raise RuntimeError("file indexing failed")
time.sleep(2)
answer = requests.post(
f"{BASE}/v1/chat/completions",
headers={**AUTH,
"X-GhostMind-Upstream-Project": slug,
"Content-Type": "application/json"},
json={"model": "gpt-4o", "messages": [{"role": "user",
"content": "Summarize the brand voice file."}]},
timeout=180,
)

Async speech generation

Speech takes ~30–120 s upstream — always use the job endpoint, never the bounded sync path for real workloads.

job = requests.post(
f"{BASE}/v1/audio/generations",
headers={**AUTH, "Content-Type": "application/json"},
json={"input": "مرحباً بك في منصتنا",
"voice": "savio_default",
"response_format": "wav"},
).json() # 202 Accepted
while True:
status = requests.get(
f"{BASE}/v1/audio/generations/{job['id']}", headers=AUTH).json()["status"]
if status == "completed":
break
if status in ("failed", "cancelled"):
raise RuntimeError(f"job {status}")
time.sleep(5)
audio = requests.get(
f"{BASE}/v1/audio/generations/{job['id']}/content", headers=AUTH)
audio.raise_for_status()
open("clip.wav", "wb").write(audio.content)
# Generated content expires after 24 h by default
# (operator-configurable via GHOSTMIND_AUDIO_JOB_TTL_HOURS).

Transcription

with open("clip.wav", "rb") as f:
result = requests.post(
f"{BASE}/v1/audio/transcriptions",
headers=AUTH,
files={"file": f},
data={"model": "whisper-1"},
).json()
print(result["text"]) # usually < 2 s for short clips

Usage

usage = requests.get(f"{BASE}/v1/usage?period=7d", headers=AUTH).json()
print(usage["total_requests"], usage["total_tokens_in"],
usage["total_tokens_out"])
# Scoped to THIS api key — never other tenants.

Next Steps