{"benches":[{"id":"artificial-analysis-intelligence-index","label":"Artificial Analysis Intelligence Index v4.3","url":"https://artificialanalysis.ai/methodology/intelligence-benchmarking","scored":28},{"id":"gdpval-aa","label":"GDPval-AA v2","url":"https://artificialanalysis.ai/evaluations/gdpval-aa","scored":28},{"id":"terminal-bench","label":"Terminal-Bench 4.0","url":"https://artificialanalysis.ai/evaluations/terminal-bench","scored":26},{"id":"agents-last-exam","label":"Agents' Last Exam","url":"https://agents-last-exam.org/leaderboard","scored":3},{"id":"design-arena-full-stack","label":"Design Arena — Full Stack","url":"https://www.designarena.ai/leaderboard/fullstack","scored":18}],"providers":[{"id":"claude-code-cli","label":"claude-code-cli","cli":"claude -p","efforts":["","low","medium","high","xhigh","max"],"fast":true,"models":7},{"id":"codex-app-server","label":"codex-app-server","cli":"codex app-server","efforts":["","low","medium","high","xhigh","max","ultra"],"fast":true,"models":16},{"id":"antigravity-cli","label":"antigravity-cli","cli":"agy","efforts":["","low","medium","high"],"fast":false,"models":5}],"how":[{"id":"key","takes":"one of `keys` below","means":"a shortlist somebody chose. the first vendor on it you are signed into answers"},{"id":"intelligence","takes":"0-10, and an optional `bench`","means":"the left edge of that board, spread over eleven rungs"},{"id":"model","takes":"`model`, optional `effort` and `fast`","means":"you already know. passed to the CLI verbatim"}],"keys":[{"id":"fast","says":"answer soonest. the quickest thing each vendor has, running hot","picks":[{"provider":"codex-app-server","model":"gpt-6-astra","effort":"low","fast":true},{"provider":"antigravity-cli","model":"gemini-3.8-flash-low","effort":"","fast":false},{"provider":"claude-code-cli","model":"claude-opus-5","effort":"low","fast":true}]},{"id":"code","says":"write and change code","picks":[{"provider":"codex-app-server","model":"gpt-6-astra","effort":"high","fast":false},{"provider":"claude-code-cli","model":"claude-opus-5","effort":"max","fast":false},{"provider":"antigravity-cli","model":"gemini-3.8-flash-high","effort":"","fast":false}]},{"id":"design","says":"make something somebody has to look at","picks":[{"provider":"claude-code-cli","model":"claude-opus-5","effort":"max","fast":false},{"provider":"codex-app-server","model":"gpt-6-astra","effort":"high","fast":false},{"provider":"antigravity-cli","model":"gemini-3.8-flash-high","effort":"","fast":false}]},{"id":"research","says":"read a lot, and be right","picks":[{"provider":"codex-app-server","model":"gpt-6-astra","effort":"high","fast":false},{"provider":"claude-code-cli","model":"claude-opus-5","effort":"max","fast":false},{"provider":"antigravity-cli","model":"gemini-3.8-flash-high","effort":"","fast":false}]},{"id":"cost","says":"spend as little as the job allows","picks":[{"provider":"codex-app-server","model":"gpt-6-astra","effort":"low","fast":false},{"provider":"antigravity-cli","model":"gemini-3.8-flash-low","effort":"","fast":false},{"provider":"claude-code-cli","model":"claude-opus-5","effort":"low","fast":false}]},{"id":"general","says":"no strong opinion. a good default","picks":[{"provider":"codex-app-server","model":"gpt-6-astra","effort":"high","fast":false},{"provider":"claude-code-cli","model":"claude-opus-5","effort":"medium","fast":false},{"provider":"antigravity-cli","model":"gemini-3.8-flash-high","effort":"","fast":false}]}],"models":28,"source":{"benchmark":"Artificial Analysis Intelligence Index v4.3, GDPval-AA v2, Terminal-Bench 4.0","url":"https://artificialanalysis.ai/models/gpt-6-astra","score":"Elo from blind pairwise judging of complete work deliverables over 220 GDPval tasks in an agentic harness, anchored so a human expert scores 1000. 95% CIs run about +/-15 to +/-27, so gaps under ~35 Elo are not separable.","price":"USD per Artificial Analysis Intelligence Index v4.3 task; null means unmeasured.","captured":"2026-09-09","secs":"Seconds per Artificial Analysis Intelligence Index v4.3 task; null means unmeasured.","provider":"Artificial Analysis","design":"Design Arena, Full Stack leaderboard, per-criteria average. Captured 2026-08-25 from https://www.designarena.ai/leaderboard/fullstack","research":"Agents' Last Exam, Score (average partial credit) for the best effort each harness publishes. Captured 2026-08-25 from https://agents-last-exam.org/leaderboard","database":"https://github.com/teamofsilicons/omnipotent/blob/main/models.json"},"fresh":true}