{"how":"intelligence","intelligence":5,"bench":"gdpval-aa","benchmark":{"label":"GDPval-AA v2","short":"gdpval","url":"https://artificialanalysis.ai/evaluations/gdpval-aa","field":"gdpval"},"pick":{"provider":"codex-app-server","model":"gpt-6-astra","effort":"high","fast":false,"run":"codex app-server --stdio · turn/start { model: \"gpt-6-astra\", effort: \"high\" }"},"ladder":{"0":{"provider":"claude-code-cli","model":"claude-haiku-4-5-20251001","effort":"","fast":false,"run":"claude -p --model claude-haiku-4-5-20251001"},"1":{"provider":"codex-app-server","model":"gpt-5.6-terra","effort":"high","fast":false,"run":"codex app-server --stdio · turn/start { model: \"gpt-5.6-terra\", effort: \"high\" }"},"2":{"provider":"codex-app-server","model":"gpt-5.6-terra","effort":"xhigh","fast":false,"run":"codex app-server --stdio · turn/start { model: \"gpt-5.6-terra\", effort: \"xhigh\" }"},"3":{"provider":"codex-app-server","model":"gpt-6-astra","effort":"medium","fast":false,"run":"codex app-server --stdio · turn/start { model: \"gpt-6-astra\", effort: \"medium\" }"},"4":{"provider":"codex-app-server","model":"gpt-6-astra","effort":"high","fast":false,"run":"codex app-server --stdio · turn/start { model: \"gpt-6-astra\", effort: \"high\" }"},"5":{"provider":"codex-app-server","model":"gpt-6-astra","effort":"high","fast":false,"run":"codex app-server --stdio · turn/start { model: \"gpt-6-astra\", effort: \"high\" }"},"6":{"provider":"codex-app-server","model":"gpt-6-astra","effort":"xhigh","fast":false,"run":"codex app-server --stdio · turn/start { model: \"gpt-6-astra\", effort: \"xhigh\" }"},"7":{"provider":"codex-app-server","model":"gpt-6-astra","effort":"max","fast":false,"run":"codex app-server --stdio · turn/start { model: \"gpt-6-astra\", effort: \"max\" }"},"8":{"provider":"claude-code-cli","model":"claude-opus-5","effort":"high","fast":false,"run":"claude -p --model claude-opus-5 --effort high"},"9":{"provider":"claude-code-cli","model":"claude-opus-5","effort":"xhigh","fast":false,"run":"claude -p --model claude-opus-5 --effort xhigh"},"10":{"provider":"claude-code-cli","model":"claude-opus-5","effort":"max","fast":false,"run":"claude -p --model claude-opus-5 --effort max"}},"skipped":[{"provider":"antigravity-cli","model":"gemini-3.8-flash-medium","effort":"","why":"no cost or time measured yet"},{"provider":"codex-app-server","model":"gpt-5.4","effort":"xhigh","why":"no cost or time measured yet"},{"provider":"antigravity-cli","model":"gemini-3.8-flash-low","effort":"","why":"no cost or time measured yet"},{"provider":"codex-app-server","model":"gpt-5.6-terra","effort":"medium","why":"no cost or time measured yet"},{"provider":"codex-app-server","model":"gpt-5.5","effort":"low","why":"no cost or time measured yet"},{"provider":"codex-app-server","model":"gpt-5.6-terra","effort":"low","why":"no cost or time measured yet"}],"caveats":["Artificial Analysis measures first-party API configurations, which may differ from the CLI harnesses used here.","Price and time use Intelligence Index v4.3 tasks. Missing measurements remain null and are excluded from numerical scales, but explicit keyword picks remain runnable.","Astra ultra is supported by the app server but has no separately measured row here; its score is not inferred from max.","Keyword shortlists are curated routing preferences. They preserve the requested Astra low/high and Flash 3.8 choices; they are not numerical benchmark rankings.","Supplemental design and research measurements are being audited separately; no previous-model scores are transferred to Astra or Flash 3.8."],"providers":["antigravity-cli","claude-code-cli","codex-app-server"],"source":{"benchmark":"Artificial Analysis Intelligence Index v4.3, GDPval-AA v2, Terminal-Bench 4.0","url":"https://artificialanalysis.ai/models/gpt-6-astra","score":"Elo from blind pairwise judging of complete work deliverables over 220 GDPval tasks in an agentic harness, anchored so a human expert scores 1000. 95% CIs run about +/-15 to +/-27, so gaps under ~35 Elo are not separable.","price":"USD per Artificial Analysis Intelligence Index v4.3 task; null means unmeasured.","captured":"2026-09-09","secs":"Seconds per Artificial Analysis Intelligence Index v4.3 task; null means unmeasured.","provider":"Artificial Analysis","design":"Design Arena, Full Stack leaderboard, per-criteria average. Captured 2026-08-25 from https://www.designarena.ai/leaderboard/fullstack","research":"Agents' Last Exam, Score (average partial credit) for the best effort each harness publishes. Captured 2026-08-25 from https://agents-last-exam.org/leaderboard","database":"https://github.com/teamofsilicons/omnipotent/blob/main/models.json"},"fresh":true}