Many questions, one request#
The state is read once and shared by every question, so put every question about an input into one request. Ten questions cost the state's tokens once, not ten times, and all ten answers come from the same read of the input.
{
"model": "decisionnode-flash-latest",
"state": "Refund my order 4471 or I will dispute the charge with my bank.",
"questions": {
"refund_request": {
"type": "truth",
"instructions": "Is the customer asking for a refund?"
},
"chargeback_threat": {
"type": "truth",
"instructions": "Does the customer threaten a chargeback?"
},
"order_id_present": {
"type": "truth",
"instructions": "Does the message include an order number?"
},
"sentiment": {
"type": "score",
"criteria": ["calm", "annoyed", "angry", "furious"]
},
"language": {
"type": "choice",
"criteria": {
"en": "English",
"de": "German",
"es": "Spanish",
"other": "another language"
}
}
}
}Many inputs, in parallel#
Each request is independent, so a backlog parallelises cleanly. Bound the concurrency so you stay under your limits (by default 40 requests per second, provisional) and retry 429s after Retry-After.
import asyncio, os, httpx
LIMIT = asyncio.Semaphore(16)
async def decide(client, state):
async with LIMIT:
r = await client.post("https://api.decisionnode.com/v1/decide", json={
"model": "decisionnode-flash-latest",
"state": state,
"questions": QUESTIONS,
})
r.raise_for_status()
return r.json()["answers"]
async def run(states):
headers = {"Authorization": f"Bearer {os.environ['DECISIONNODE_API_KEY']}"}
async with httpx.AsyncClient(headers=headers, timeout=10) as client:
return await asyncio.gather(*(decide(client, s) for s in states))