Skip to content

API Reference

Examples

End-to-end API examples covering comparisons, evaluations, benchmarks, and RAG.

All examples use Python with the requests library and the polling helper from Async jobs.

import time, requests

API_KEY = "YOUR_API_KEY"
BASE = "https://api.llmprover.pysolvr.com"
H = {"Authorization": API_KEY, "Content-Type": "application/json"}

def poll(job_id, timeout=300):
    deadline = time.time() + timeout
    while time.time() < deadline:
        time.sleep(2)
        r = requests.get(f"{BASE}/jobs/{job_id}", headers=H).json()
        s = r["data"]["status"]
        if s == "complete": return r["data"]["result"]
        if s == "failed": raise RuntimeError(r["data"].get("error"))
    raise TimeoutError("Job timed out")

Compare two models with a system prompt, inference parameters, and RAG context.

# 1. Create a RAG store
store = requests.post(f"{BASE}/rag/stores", headers=H, json={
    "name": "Product docs",
    "store_type": "general"
}).json()["data"]
store_id = store["store_id"]

# 2. Upload a file to the store
import os
filename = "product-docs.pdf"
size = os.path.getsize(filename)

url_resp = requests.post(f"{BASE}/files/upload-url", headers=H, json={
    "filename": filename,
    "content_type": "application/pdf",
    "purpose": "rag_context",
    "size_bytes": size,
    "store_id": store_id
}).json()["data"]

# 3. PUT the file directly to S3
with open(filename, "rb") as f:
    requests.put(url_resp["upload_url"],
                 data=f,
                 headers={"Content-Type": "application/pdf"})

# 4. Confirm upload to trigger ingestion
file_id = url_resp["file_id"]
requests.post(f"{BASE}/files/{file_id}/confirm", headers=H)

# 5. Wait for ingestion to complete
import time
for _ in range(30):
    time.sleep(5)
    stores = requests.get(f"{BASE}/rag/stores", headers=H).json()["data"]["stores"]
    s = next(s for s in stores if s["store_id"] == store_id)
    if s["sync_status"] == "ready":
        break

# 6. Run a comparison with system prompt, params, and RAG context
resp = requests.post(f"{BASE}/compare", headers=H, json={
    "prompt": "What is the return policy for damaged items?",
    "models": {
        "openai": "gpt-4o",
        "anthropic": "claude-3-5-sonnet-20241022"
    },
    "system_prompt_id": "sp_abc123",   # saved system prompt ID from dashboard
    "store_id": store_id,
    "params": {
        "temperature": 0.2,
        "max_tokens": 512
    }
}).json()

result = poll(resp["data"]["job_id"])
for r in result["results"]:
    print(f"{r['provider']}: {r['response_text'][:100]}...")
    print(f"  cost: ${r['cost_usd']:.6f}  latency: {r['latency_ms']}ms")

Example 2 – Evaluation with rubric

Score responses against a rubric. The rubric must be created in the dashboard first – note its rubric_id.

resp = requests.post(f"{BASE}/evaluate", headers=H, json={
    "prompt": "A customer says their order arrived damaged. Write a support reply.",
    "models": {
        "openai": "gpt-4o",
        "anthropic": "claude-3-5-sonnet-20241022"
    },
    "rubric_id": "rubric_abc123",   # from dashboard
    "params": {"temperature": 0.3}
}).json()

result = poll(resp["data"]["job_id"])
for r in result["results"]:
    print(f"{r['provider']} quality score: {r.get('quality_score')}")
    for score in (r.get("scores") or []):
        print(f"  {score['criterion']}: {score['score']} -- {score['reasoning'][:80]}")

Example 3 – Evaluation with gold standard

Score responses against a known correct answer instead of a rubric.

resp = requests.post(f"{BASE}/evaluate", headers=H, json={
    "prompt": "What is the capital of France?",
    "models": {"openai": "gpt-4o-mini"},
    "expected_output": "Paris"
}).json()

result = poll(resp["data"]["job_id"])
print(result["results"][0]["quality_score"])

Example 4 – Create and run a benchmark

Create a daily benchmark with thresholds, then trigger a manual run.

# Create the benchmark
bench = requests.post(f"{BASE}/benchmarks", headers=H, json={
    "name": "Support reply quality",
    "prompt": "A customer says their order arrived damaged. Write a support reply.",
    "models": {
        "openai": "gpt-4o",
        "anthropic": "claude-3-5-sonnet-20241022"
    },
    "rubric_id": "rubric_abc123",
    "schedule": "daily",
    "thresholds": {
        "quality_score": {"min": 80},
        "cost_usd": {"max": 0.05},
        "latency_ms": {"max": 5000}
    }
}).json()["data"]

suite_id = bench["suite"]["suite_id"]

# Trigger a manual run
run_resp = requests.post(f"{BASE}/benchmarks/{suite_id}/run", headers=H).json()
result = poll(run_resp["data"]["job_id"])

# Check per-model metrics
for m in result.get("model_metrics", []):
    status = "PASS" if m.get("threshold_pass") else "FAIL"
    print(f"{m['provider']}: quality={m.get('quality_score')} cost=${m.get('cost_usd'):.6f} [{status}]")

Example 5 – List run history and check for anomalies

runs = requests.get(
    f"{BASE}/benchmarks/{suite_id}/runs",
    headers=H,
    params={"limit": 10}
).json()["data"]["runs"]

for run in runs:
    flags = run.get("anomaly_flags", [])
    print(f"Run {run['run_id'][:8]}: {len(flags)} anomaly flags")
    for flag in flags:
        print(f"  {flag}")

Example 6 – Swap a model on an existing benchmark

# Switch openai from gpt-4o to gpt-4o-mini
requests.patch(f"{BASE}/benchmarks/{suite_id}", headers=H, json={
    "models": {"openai": "gpt-4o-mini"}
})

# Add a new provider
requests.patch(f"{BASE}/benchmarks/{suite_id}", headers=H, json={
    "providers": ["openai", "anthropic", "grok"]
})

What’s next