{"name":"TryAii-Bench API","version":"1.0.0","description":"A read-only REST API serving LLM benchmark data from Supabase (Postgres). Data covers 330+ LLM models across pricing, capabilities, speed, and benchmark scores from 8+ independent public sources. Data is refreshed automatically every 6 hours.","authentication":{"method":"API key via x-api-key header","note":"Authentication is disabled only in a local environment (ENV unset or local/dev/development/test) with no API_KEY set. Anywhere else an empty API_KEY is a fatal misconfiguration: the process refuses to start."},"endpoints":{"GET /":"This page — API overview, endpoint list, and usage guide.","GET /summary":"High-level stats: row counts, unique models, last refresh time.","GET /models":"All model metadata (name, description, context length, modality, etc.).","GET /models/list":"Lightweight list of all model IDs and display names, plus per-model archetype and indexable flags (NULL until a summary row exists).","GET /models/{model_id}":"Metadata for a single model.","GET /models/{model_id}/full":"Combined view: model + pricing + capabilities + benchmarks + speed + cache thresholds + summary.","GET /models/{model_id}/summary":"Generated editorial summary: page blurb, SERP meta description (<=155 chars), archetype, fact pack, indexable flag. Regenerated every 6 hours. Add ?lean=true to drop the fact pack and row bookkeeping.","GET /models/{model_id}/cache-thresholds":"Per-model prompt-cache activation factors, one row per serving platform (same rows as /cache/thresholds/{model_id}). Add ?lean=true for provider_name + min_input_tokens + notes only.","GET /pricing":"Cost per 1M tokens (prompt & completion) for all models.","GET /pricing/{model_id}":"Pricing for a single model.","GET /capabilities":"Capability flags: function calling, JSON mode, vision, context length, etc.","GET /capabilities/{model_id}":"Capabilities for a single model.","GET /benchmarks/static":"Published benchmark scores (IFEval, BBH, MATH, GPQA, MMLU-Pro, HumanEval, etc.).","GET /benchmarks/static/{model_id}":"Static benchmark scores for a single model.","GET /benchmarks/live":"Live benchmark scores (Chatbot Arena Elo, LiveBench, MT-Bench).","GET /benchmarks/live/{model_id}":"Live benchmark scores for a single model.","GET /benchmarks/meta":"Benchmark registry: category, difficulty tier, active/frozen status, score direction, and lineage (family / successor_of) per benchmark label. Filters: ?difficulty_tier=, ?status=, ?family=.","GET /speed":"Throughput (tokens/sec), latency (TTFT, avg) per model per provider.","GET /speed/{model_id}":"Speed metrics for a single model across all providers.","GET /cache/thresholds":"Prompt-cache activation factors per (model, serving platform): minimum input tokens before caching kicks in, enablement mode, read/write price multipliers, TTL. Filters: ?model_id=, ?provider=.","GET /cache/thresholds/{model_id}":"Cache thresholds for a single model across all serving platforms.","GET /training-queries":"Curated DRE centroid training queries per benchmark (shallow list; no query contents).","GET /training-queries/{benchmark}":"Full training-query payload for a benchmark, including per-query source URLs.","GET /admin/refresh":"Clear in-memory cache to force fresh data from database."},"query_parameters":{"model_id":"Filter by exact model_id (e.g., ?model_id=openai/gpt-4o)","benchmark":"Filter benchmark datasets by benchmark name (e.g., ?benchmark=IFEval)","provider":"Filter speed / cache-threshold data by provider name (e.g., ?provider=Together, ?provider=Anthropic)","min_score":"Minimum score filter (e.g., ?min_score=0.8)","max_score":"Maximum score filter (e.g., ?max_score=0.95)","sort":"Sort by any field name (e.g., ?sort=score)","order":"Sort direction: asc (default) or desc (e.g., ?order=desc)","limit":"Maximum number of results (1-10000, e.g., ?limit=50)","offset":"Skip N results for pagination (e.g., ?offset=50)","lean":"On /models/{id}/summary and /models/{id}/cache-thresholds only: return the presentation subset of each row (e.g., ?lean=true)"},"examples":{"curl":["curl -H \"x-api-key: YOUR_KEY\" http://localhost:8000/models/list","curl http://localhost:8000/benchmarks/static?benchmark=IFEval&sort=score&order=desc&limit=10","curl http://localhost:8000/models/openai/gpt-4o/full","curl http://localhost:8000/speed?provider=Together&sort=tokens_per_second&order=desc"],"python":["import requests","r = requests.get(\"http://localhost:8000/models/list\", headers={\"x-api-key\": \"YOUR_KEY\"})","models = r.json()[\"data\"]"],"javascript":["const res = await fetch(\"http://localhost:8000/summary\", { headers: { \"x-api-key\": \"YOUR_KEY\" } });","const data = await res.json();"]},"data_sources":["OpenRouter API (models, pricing, speed)","HuggingFace Open LLM Leaderboard v2 (static benchmarks)","PapersWithCode Archive (HumanEval, MMLU, GSM8K, DROP)","SWE-bench GitHub (SWE-bench Verified & Lite)","Arena AI (Chatbot Arena Elo)","MT-Bench (LMSYS)","LiveBench (HuggingFace)","Salt Technologies (LLM comparison)","Training Query Seed Loader (curated JSON, training_queries)","Cache Threshold Seed Loader (curated JSON from the 2026-07 cache_providers research, cache_thresholds)","Model Summary Builder (derived editorial prose from model_summary.py, model_summaries)"],"docs_url":"/docs","openapi_url":"/openapi.json"}