{"how":"key","key":"code","says":"write and change code","pick":{"provider":"codex-app-server","model":"gpt-6-astra","effort":"high","fast":false,"run":"codex app-server --stdio · turn/start { model: \"gpt-6-astra\", effort: \"high\" }"},"shortlist":[{"provider":"codex-app-server","model":"gpt-6-astra","effort":"high","fast":false,"run":"codex app-server --stdio · turn/start { model: \"gpt-6-astra\", effort: \"high\" }","available":true},{"provider":"claude-code-cli","model":"claude-opus-5","effort":"max","fast":false,"run":"claude -p --model claude-opus-5 --effort max","available":true},{"provider":"antigravity-cli","model":"gemini-3.8-flash-high","effort":"","fast":false,"run":"agy --model gemini-3.8-flash-high --print","available":true}],"providers":["antigravity-cli","claude-code-cli","codex-app-server"],"source":{"benchmark":"Artificial Analysis Intelligence Index v4.3, GDPval-AA v2, Terminal-Bench 4.0","url":"https://artificialanalysis.ai/models/gpt-6-astra","score":"Elo from blind pairwise judging of complete work deliverables over 220 GDPval tasks in an agentic harness, anchored so a human expert scores 1000. 95% CIs run about +/-15 to +/-27, so gaps under ~35 Elo are not separable.","price":"USD per Artificial Analysis Intelligence Index v4.3 task; null means unmeasured.","captured":"2026-09-09","secs":"Seconds per Artificial Analysis Intelligence Index v4.3 task; null means unmeasured.","provider":"Artificial Analysis","design":"Design Arena, Full Stack leaderboard, per-criteria average. Captured 2026-08-25 from https://www.designarena.ai/leaderboard/fullstack","research":"Agents' Last Exam, Score (average partial credit) for the best effort each harness publishes. Captured 2026-08-25 from https://agents-last-exam.org/leaderboard","database":"https://github.com/teamofsilicons/omnipotent/blob/main/models.json"},"fresh":true}