API Reference
Examples
End-to-end API examples covering comparisons, evaluations, benchmarks, and RAG.
All examples use Python with the requests library and the polling helper from Async jobs.
import time, requests
API_KEY = "YOUR_API_KEY"
BASE = "https://api.llmprover.pysolvr.com"
H = {"Authorization": API_KEY, "Content-Type": "application/json"}
def poll(job_id, timeout=300):
deadline = time.time() + timeout
while time.time() < deadline:
time.sleep(2)
r = requests.get(f"{BASE}/jobs/{job_id}", headers=H).json()
s = r["data"]["status"]
if s == "complete": return r["data"]["result"]
if s == "failed": raise RuntimeError(r["data"].get("error"))
raise TimeoutError("Job timed out")
Example 1 – Full-featured comparison
Compare two models with a system prompt, inference parameters, and RAG context.
# 1. Create a RAG store
store = requests.post(f"{BASE}/rag/stores", headers=H, json={
"name": "Product docs",
"store_type": "general"
}).json()["data"]
store_id = store["store_id"]
# 2. Upload a file to the store
import os
filename = "product-docs.pdf"
size = os.path.getsize(filename)
url_resp = requests.post(f"{BASE}/files/upload-url", headers=H, json={
"filename": filename,
"content_type": "application/pdf",
"purpose": "rag_context",
"size_bytes": size,
"store_id": store_id
}).json()["data"]
# 3. PUT the file directly to S3
with open(filename, "rb") as f:
requests.put(url_resp["upload_url"],
data=f,
headers={"Content-Type": "application/pdf"})
# 4. Confirm upload to trigger ingestion
file_id = url_resp["file_id"]
requests.post(f"{BASE}/files/{file_id}/confirm", headers=H)
# 5. Wait for ingestion to complete
import time
for _ in range(30):
time.sleep(5)
stores = requests.get(f"{BASE}/rag/stores", headers=H).json()["data"]["stores"]
s = next(s for s in stores if s["store_id"] == store_id)
if s["sync_status"] == "ready":
break
# 6. Run a comparison with system prompt, params, and RAG context
resp = requests.post(f"{BASE}/compare", headers=H, json={
"prompt": "What is the return policy for damaged items?",
"models": {
"openai": "gpt-4o",
"anthropic": "claude-3-5-sonnet-20241022"
},
"system_prompt_id": "sp_abc123", # saved system prompt ID from dashboard
"store_id": store_id,
"params": {
"temperature": 0.2,
"max_tokens": 512
}
}).json()
result = poll(resp["data"]["job_id"])
for r in result["results"]:
print(f"{r['provider']}: {r['response_text'][:100]}...")
print(f" cost: ${r['cost_usd']:.6f} latency: {r['latency_ms']}ms")
Example 2 – Evaluation with rubric
Score responses against a rubric. The rubric must be created in the dashboard first – note its rubric_id.
resp = requests.post(f"{BASE}/evaluate", headers=H, json={
"prompt": "A customer says their order arrived damaged. Write a support reply.",
"models": {
"openai": "gpt-4o",
"anthropic": "claude-3-5-sonnet-20241022"
},
"rubric_id": "rubric_abc123", # from dashboard
"params": {"temperature": 0.3}
}).json()
result = poll(resp["data"]["job_id"])
for r in result["results"]:
print(f"{r['provider']} quality score: {r.get('quality_score')}")
for score in (r.get("scores") or []):
print(f" {score['criterion']}: {score['score']} -- {score['reasoning'][:80]}")
Example 3 – Evaluation with gold standard
Score responses against a known correct answer instead of a rubric.
resp = requests.post(f"{BASE}/evaluate", headers=H, json={
"prompt": "What is the capital of France?",
"models": {"openai": "gpt-4o-mini"},
"expected_output": "Paris"
}).json()
result = poll(resp["data"]["job_id"])
print(result["results"][0]["quality_score"])
Example 4 – Create and run a benchmark
Create a daily benchmark with thresholds, then trigger a manual run.
# Create the benchmark
bench = requests.post(f"{BASE}/benchmarks", headers=H, json={
"name": "Support reply quality",
"prompt": "A customer says their order arrived damaged. Write a support reply.",
"models": {
"openai": "gpt-4o",
"anthropic": "claude-3-5-sonnet-20241022"
},
"rubric_id": "rubric_abc123",
"schedule": "daily",
"thresholds": {
"quality_score": {"min": 80},
"cost_usd": {"max": 0.05},
"latency_ms": {"max": 5000}
}
}).json()["data"]
suite_id = bench["suite"]["suite_id"]
# Trigger a manual run
run_resp = requests.post(f"{BASE}/benchmarks/{suite_id}/run", headers=H).json()
result = poll(run_resp["data"]["job_id"])
# Check per-model metrics
for m in result.get("model_metrics", []):
status = "PASS" if m.get("threshold_pass") else "FAIL"
print(f"{m['provider']}: quality={m.get('quality_score')} cost=${m.get('cost_usd'):.6f} [{status}]")
Example 5 – List run history and check for anomalies
runs = requests.get(
f"{BASE}/benchmarks/{suite_id}/runs",
headers=H,
params={"limit": 10}
).json()["data"]["runs"]
for run in runs:
flags = run.get("anomaly_flags", [])
print(f"Run {run['run_id'][:8]}: {len(flags)} anomaly flags")
for flag in flags:
print(f" {flag}")
Example 6 – Swap a model on an existing benchmark
# Switch openai from gpt-4o to gpt-4o-mini
requests.patch(f"{BASE}/benchmarks/{suite_id}", headers=H, json={
"models": {"openai": "gpt-4o-mini"}
})
# Add a new provider
requests.patch(f"{BASE}/benchmarks/{suite_id}", headers=H, json={
"providers": ["openai", "anthropic", "grok"]
})
What’s next
- Gotchas – common integration mistakes
- Pagination – handling large result sets
- Changelog – API version history