{
  "retrieved": "2026-09-08",
  "source_family": "Artificial Analysis live comparison tables, Intelligence Index v4.3",
  "efforts": ["low", "medium", "high", "xhigh", "max"],
  "models": [
    {"name":"Astra", "full_name":"GPT-6 Astra", "tb":[42,49,54,60,59], "scicode":[54,54,55,56,56], "index":[46,50,51,53,53], "output_k":[4,10,12,17,27], "reasoning_k":[0.938,3,5,9,17], "cost":[0.82,1.54,1.72,2.31,3.26], "latency_s":[2.55,5.42,44.97,131.86,310.77], "decode_s":[80.71,173.56,206.47,304.65,456.51], "response500_s":[11.72,14.52,53.75,140.67,319.03]},
    {"name":"Fable 5.1", "full_name":"Claude Fable 5.1, adaptive reasoning, default fallback", "tb":[40,45,52,55,52], "scicode":[57,56,59,61,63], "index":[47,49,51,53,53], "output_k":[22,28,38,61,78], "reasoning_k":[8,12,19,34,47], "cost":[2.37,2.98,3.91,5.98,7.63], "latency_s":[6.04,7.83,23.85,99.86,275.52], "decode_s":[246.52,321.76,427.48,627.65,731.44], "response500_s":[15.60,17.01,32.77,108.48,282.81]}
  ],
  "sources": [
    ["1", "AA: low-effort comparison", "https://artificialanalysis.ai/models/comparisons/gpt-6-astra-low-vs-claude-fable-5-1-low"],
    ["2", "AA: medium-effort comparison", "https://artificialanalysis.ai/models/comparisons/gpt-6-astra-medium-vs-claude-fable-5-1-medium"],
    ["3", "AA: high-effort comparison", "https://artificialanalysis.ai/models/comparisons/gpt-6-astra-high-vs-claude-fable-5-1-high"],
    ["4", "AA: xhigh-effort comparison", "https://artificialanalysis.ai/models/comparisons/gpt-6-astra-xhigh-vs-claude-fable-5-1-xhigh"],
    ["5", "AA: max-effort comparison", "https://artificialanalysis.ai/models/comparisons/gpt-6-astra-vs-claude-fable-5-1"],
    ["6", "AA: Terminal-Bench v4.0", "https://artificialanalysis.ai/evaluations/terminalbench-v4-0"],
    ["7", "AA: intelligence methodology", "https://artificialanalysis.ai/methodology/intelligence-benchmarking"],
    ["8", "OpenAI: Astra model and pricing", "https://developers.openai.com/api/docs/models/gpt-6-astra"],
    ["9", "Anthropic: Fable and pricing", "https://www.anthropic.com/claude/fable"],
    ["10", "Anthropic: Fable 5.1 launch", "https://www.anthropic.com/claude-fable-and-mythos-5-1"],
    ["11", "AA: Fable launch evaluation and fallback", "https://artificialanalysis.ai/articles/claude-fable-5-1"],
    ["12", "Firecrawl: small-sample API measurements", "https://www.firecrawl.dev/blog/is-fable-5-1-cheaper-than-fable-5"]
  ],
  "notes": [
    "Coding scores are whole-percent values as displayed in the five comparison tables, not reconstructed trial counts.",
    "Output and reasoning tokens per task, and USD per task, are weighted Intelligence Index v4.3 aggregates, not coding-only measurements.",
    "Reasoning is a subset of total output. Token values marked k are rounded source values. Do not add reasoning to total output for billing.",
    "Latency is the comparison table's time to first answer token, from a separate API performance workload; not time to a verified code patch.",
    "decode_s is AA's calculated weighted decode time, excluding TTFT and overhead. It is not measured task wall time.",
    "Connecting lines are visual guides between ordinal effort settings, not fitted continuous response functions.",
    "AA leaderboard precise top scores are Astra xhigh 59.6%, Astra max 59.1%, Fable xhigh 55.1%; all plots retain the comparison tables' uniform whole-percent precision.",
    "Comparison and model pages differ in rounded suite totals (Fable xhigh 121M vs 120M and max 188M vs 190M); suite totals are excluded from the curves.",
    "No fresh model API benchmark was run. This is a synthesis of published measurements."
  ]
}
