python
# before client = OpenAI(api_key=OPENAI_KEY) # after client = OpenAI( base_url="https://gw.lematev.dev/v1", api_key=LEMATEV_KEY, )
python
client.chat.completions.create(
model="gpt-5",
messages=messages,
extra_headers={
"x-lematev-agent": "support-triage",
"x-lematev-run": run_id,
},
)
- x-lematev-agent
- x-lematev-run
python
requests.post(
"https://gw.lematev.dev/v1/feedback",
headers={"x-lematev-key": LEMATEV_KEY},
json={"call_id": call_id, "success": False},
)
python
stream = client.chat.completions.create(
model="gpt-5", messages=messages, stream=True,
extra_headers={"x-lematev-agent": "support-triage"},
)
for chunk in stream:
print(chunk.choices[0].delta.content or "", end="")
reconciliation
your agents $ 81.72 + counterfactual replays $ 2.34 ────────────────────────────────────── = what your provider bills $ 84.06 baseline (what you would have paid) $127.96 − savings $ 46.24 ────────────────────────────────────── = your agents $ 81.72
webhook payload
{
"event": "budget.exceeded",
"workspace": "Production",
"agent": "support-triage",
"spent_usd": 412.80,
"budget_usd": 400.00
}
json
{
"error": {
"message": "3 identical calls detected in this run…",
"type": "lematev_execution_stopped",
"code": "identical_repeat"
},
"lematev": {
"run_id": "run_9f2c…",
"estimated_usd_saved": 0.58
}
}
python
except openai.APIStatusError as e:
body = e.response.json()
# 409 is ours: the run was stopped, not the provider failing.
if e.status_code == 409:
alert("agent looping", body["lematev"]["run_id"])
elif e.status_code == 402:
alert("limit reached", body["error"]["code"])
else:
raise
your server
token = requests.post(
"https://gw.lematev.dev/v1/client-tokens",
headers={"authorization": f"Bearer {LEMATEV_KEY}"},
json={"agent": "chat-mobile",
"ttl_seconds": 600,
"max_calls": 20},
).json()["token"] # lmc_… — hand this to the phone