{
  "name": "Aeon Model Economics Lab dataset",
  "version": "2026-08-14",
  "updated": "2026-08-14",
  "license": "Aeon research compilation. Source benchmark terms continue to apply.",
  "methodology": "The default master plot shows every new release plus the measured shortlist. The y-axis displays each point's published coding score in its native suite, the x-axis displays published or explicitly proxied task cost, and marker area displays published or proxied tokens. Filled circles are comparable Artificial Analysis Coding Agent Index measurements. Diamonds are provisional cross-suite, mixed-basis, or family-proxy points whose scoreMetric, costBasis, tokenMetric, and provenance must be inspected before comparison. Popular, Value, and All/history views retain the broader dataset.",
  "sources": [
    {
      "label": "OpenAI GPT-5.6 launch and pricing",
      "url": "https://openai.com/index/gpt-5-6/"
    },
    {
      "label": "Artificial Analysis Coding Agent Index",
      "url": "https://artificialanalysis.ai/agents/coding-agents"
    },
    {
      "label": "Artificial Analysis GPT-5.6 analysis",
      "url": "https://artificialanalysis.ai/articles/gpt-5-6-has-landed"
    },
    {
      "label": "OpenAI GPT-5.5 launch",
      "url": "https://openai.com/index/introducing-gpt-5-5/"
    },
    {
      "label": "xAI Grok 4.5 announcement",
      "url": "https://x.ai/news/grok-4-5"
    },
    {
      "label": "Anthropic Claude Fable 5",
      "url": "https://www.anthropic.com/claude/fable"
    },
    {
      "label": "Anthropic effort documentation",
      "url": "https://platform.claude.com/docs/en/build-with-claude/effort"
    },
    {
      "label": "Z.ai GLM-5.2 official release",
      "url": "https://z.ai/blog/glm-5.2"
    },
    {
      "label": "Artificial Analysis Claude Opus 5 analysis",
      "url": "https://artificialanalysis.ai/articles/opus-5"
    },
    {
      "label": "Meta Muse Spark 1.1 release",
      "url": "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/"
    },
    {
      "label": "Artificial Analysis Muse Spark 1.1 measurements",
      "url": "https://artificialanalysis.ai/models/muse-spark-1-1/"
    },
    {
      "label": "Alibaba Model Studio Qwen3.8-Max-Preview listing",
      "url": "https://qwenlm.github.io/qwen-code-docs/en/blog/updates/weekly-update-2026-07-23/"
    },
    {
      "label": "Kimi K3 official launch and pricing",
      "url": "https://www.kimi.com/blog/kimi-k3"
    },
    {
      "label": "Artificial Analysis Kimi K3 measurements",
      "url": "https://artificialanalysis.ai/models/kimi-k3"
    },
    {
      "label": "CursorBench 3.2 evaluations",
      "url": "https://cursor.com/evals"
    },
    {
      "label": "Google Gemini 3.7 Flash release notes",
      "url": "https://ai.google.dev/gemini-api/docs/changelog"
    },
    {
      "label": "xAI Grok 4.6 release notes",
      "url": "https://docs.x.ai/developers/release-notes"
    },
    {
      "label": "DeepSeek V4 Pro update",
      "url": "https://api-docs.deepseek.com/updates#deepseek-v4-pro-update"
    },
    {
      "label": "DeepSeek V4 Pro models and pricing",
      "url": "https://api-docs.deepseek.com/quick_start/pricing/"
    },
    {
      "label": "Artificial Analysis DeepSeek V4 Flash 0731",
      "url": "https://artificialanalysis.ai/models/deepseek-v4-flash"
    },
    {
      "label": "Meta Muse Code and Muse Spark 1.2 release",
      "url": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2"
    },
    {
      "label": "Agentic coding token study",
      "url": "https://arxiv.org/abs/2604.22750"
    }
  ],
  "readerViews": [
    {
      "id": "current",
      "label": "Current releases",
      "description": "Every current release by default: measured anchors plus clearly marked cross-suite and proxy diamonds.",
      "modelIds": [
        "sol",
        "opus5",
        "kimi-k3",
        "glm52",
        "grok46",
        "gemini37",
        "deepseek-v4-pro-0813",
        "deepseek-v4-flash-0731",
        "muse-spark-12"
      ]
    },
    {
      "id": "popular",
      "label": "Popular models",
      "description": "Widely recognized current and previous-generation models readers still compare.",
      "modelIds": [
        "sol",
        "opus5",
        "fable",
        "gpt55",
        "grok",
        "kimi-k3",
        "glm52",
        "grok46",
        "gemini37"
      ]
    },
    {
      "id": "value",
      "label": "Value challengers",
      "description": "Lower-cost and independent alternatives that can change a routing decision.",
      "modelIds": [
        "terra",
        "luna",
        "grok",
        "kimi-k3",
        "glm52",
        "muse-spark",
        "deepseek-v4-pro-0813",
        "deepseek-v4-flash-0731",
        "muse-spark-12"
      ]
    },
    {
      "id": "all",
      "label": "All / history",
      "description": "Every retained model family. Turn on all effort levels for the complete measured, modeled, and provisional dataset.",
      "modelIds": [
        "sol",
        "terra",
        "luna",
        "gpt55",
        "grok",
        "fable",
        "glm52",
        "kimi-k3",
        "opus5",
        "muse-spark",
        "grok46",
        "gemini37",
        "deepseek-v4-pro-0813",
        "deepseek-v4-flash-0731",
        "muse-spark-12"
      ]
    }
  ],
  "series": [
    {
      "id": "sol",
      "name": "GPT-5.6 Sol",
      "shortName": "Sol",
      "provider": "OpenAI",
      "color": "#E3B341",
      "recommendedEffort": "Max",
      "readerLabel": "Current flagship",
      "points": [
        {
          "effort": "Low",
          "score": 53.6056,
          "outputTokens": 10620,
          "cost": 1.717516,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "Medium",
          "score": 60.6144,
          "outputTokens": 19115,
          "cost": 2.991342,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "High",
          "score": 64.1096,
          "outputTokens": 28199,
          "cost": 4.144239,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "XHigh",
          "score": 65.0924,
          "outputTokens": 38288,
          "cost": 5.235289,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "Max",
          "score": 66.5698,
          "outputTokens": 54860,
          "cost": 7.08351,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        }
      ]
    },
    {
      "id": "terra",
      "name": "GPT-5.6 Terra",
      "shortName": "Terra",
      "provider": "OpenAI",
      "color": "#4FB6C2",
      "recommendedEffort": "Max",
      "readerLabel": "Value challenger",
      "points": [
        {
          "effort": "Low",
          "score": 36.7276,
          "outputTokens": 8053,
          "cost": 0.484324,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "Medium",
          "score": 47.7977,
          "outputTokens": 16002,
          "cost": 0.901226,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "High",
          "score": 55.7927,
          "outputTokens": 31662,
          "cost": 1.58974,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "XHigh",
          "score": 57.0736,
          "outputTokens": 40160,
          "cost": 1.895132,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "Max",
          "score": 62.2804,
          "outputTokens": 60732,
          "cost": 2.757235,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        }
      ]
    },
    {
      "id": "luna",
      "name": "GPT-5.6 Luna",
      "shortName": "Luna",
      "provider": "OpenAI",
      "color": "#42B883",
      "recommendedEffort": "High",
      "readerLabel": "Cost leader",
      "points": [
        {
          "effort": "Low",
          "score": 25.0834,
          "outputTokens": 6699,
          "cost": 0.208556,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "Medium",
          "score": 42.4069,
          "outputTokens": 15198,
          "cost": 0.473896,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "High",
          "score": 51.4167,
          "outputTokens": 32273,
          "cost": 0.958663,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "XHigh",
          "score": 54.6701,
          "outputTokens": 47114,
          "cost": 1.256385,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "Max",
          "score": 58.6598,
          "outputTokens": 64941,
          "cost": 1.566957,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        }
      ]
    },
    {
      "id": "gpt55",
      "name": "GPT-5.5",
      "shortName": "GPT-5.5",
      "provider": "OpenAI",
      "color": "#CBD5E1",
      "recommendedEffort": "XHigh",
      "readerLabel": "Popular legacy",
      "points": [
        {
          "effort": "Low",
          "score": 47,
          "outputTokens": 8100,
          "cost": 1.45,
          "evidence": "modeled",
          "basis": "Estimated from measured GPT-5.5 medium and xhigh runs plus the current GPT-5.6 effort curves.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "Medium",
          "score": 54.3587,
          "outputTokens": 17196,
          "cost": 2.754519,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "High",
          "score": 58,
          "outputTokens": 26000,
          "cost": 3.75,
          "evidence": "modeled",
          "basis": "Monotone interpolation between measured GPT-5.5 medium and xhigh runs.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "XHigh",
          "score": 61.4851,
          "outputTokens": 37947,
          "cost": 5.069576,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        }
      ]
    },
    {
      "id": "grok",
      "name": "Grok 4.5",
      "shortName": "Grok",
      "provider": "xAI",
      "color": "#EF7C5A",
      "recommendedEffort": "High",
      "readerLabel": "Popular legacy",
      "points": [
        {
          "effort": "Low",
          "score": 48,
          "outputTokens": 12700,
          "cost": 0.78,
          "evidence": "modeled",
          "basis": "Estimated by applying the current GPT-5.6 low-to-high effort shape to the measured Grok high anchor.",
          "sourceUrl": "https://x.ai/news/grok-4-5"
        },
        {
          "effort": "Medium",
          "score": 58,
          "outputTokens": 25200,
          "cost": 1.56,
          "evidence": "modeled",
          "basis": "Estimated by applying the current GPT-5.6 medium-to-high effort shape to the measured Grok high anchor.",
          "sourceUrl": "https://x.ai/news/grok-4-5"
        },
        {
          "effort": "High",
          "score": 64.4392,
          "outputTokens": 39966,
          "cost": 2.594053,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        }
      ]
    },
    {
      "id": "fable",
      "name": "Claude Fable 5",
      "shortName": "Fable",
      "provider": "Anthropic",
      "color": "#B99AF8",
      "recommendedEffort": "Max",
      "readerLabel": "Popular model",
      "points": [
        {
          "effort": "Low",
          "score": 24.3,
          "outputTokens": 33340,
          "cost": 3.27,
          "evidence": "modeled",
          "basis": "Anthropic effort-curve shape normalized to the current measured Fable max Coding Agent Index anchor.",
          "sourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort"
        },
        {
          "effort": "Medium",
          "score": 37.9,
          "outputTokens": 43920,
          "cost": 4.32,
          "evidence": "modeled",
          "basis": "Anthropic effort-curve shape normalized to the current measured Fable max Coding Agent Index anchor.",
          "sourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort"
        },
        {
          "effort": "High",
          "score": 51.2,
          "outputTokens": 61350,
          "cost": 6.03,
          "evidence": "modeled",
          "basis": "Anthropic effort-curve shape normalized to the current measured Fable max Coding Agent Index anchor.",
          "sourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort"
        },
        {
          "effort": "XHigh",
          "score": 62.3,
          "outputTokens": 76620,
          "cost": 7.53,
          "evidence": "modeled",
          "basis": "Anthropic effort-curve shape normalized to the current measured Fable max Coding Agent Index anchor.",
          "sourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort"
        },
        {
          "effort": "Max",
          "score": 65.847,
          "outputTokens": 73567,
          "cost": 11.710527,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026. The tested configuration used fallback.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        }
      ]
    },
    {
      "id": "glm52",
      "name": "GLM-5.2",
      "shortName": "GLM-5.2",
      "provider": "Z.ai",
      "color": "#5AA7FF",
      "recommendedEffort": "Max",
      "readerLabel": "Value challenger",
      "points": [
        {
          "effort": "Max",
          "score": 43.1778,
          "outputTokens": 40310,
          "cost": 6.5111,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026. Artificial Analysis displays GLM-5.2 without an effort suffix; Max identifies the tested reasoning configuration, not a measured effort curve.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        }
      ]
    },
    {
      "id": "kimi-k3",
      "name": "Kimi K3",
      "shortName": "Kimi K3",
      "provider": "Kimi (Moonshot AI)",
      "color": "#67D7E8",
      "recommendedEffort": "Max",
      "readerLabel": "Value challenger",
      "points": [
        {
          "effort": "Max",
          "score": 61.3354,
          "outputTokens": 88806,
          "cost": 3.175187,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        }
      ]
    },
    {
      "id": "opus5",
      "name": "Claude Opus 5",
      "shortName": "Opus 5",
      "provider": "Anthropic",
      "color": "#F08A68",
      "recommendedEffort": "XHigh",
      "readerLabel": "Quality leader",
      "points": [
        {
          "effort": "Low",
          "score": 56.7953,
          "outputTokens": 22281,
          "cost": 2.178461,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "Medium",
          "score": 61.9194,
          "outputTokens": 29632,
          "cost": 3.143624,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "High",
          "score": 63.3731,
          "outputTokens": 35447,
          "cost": 3.797452,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "XHigh",
          "score": 66.7438,
          "outputTokens": 72794,
          "cost": 8.234594,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        },
        {
          "effort": "Max",
          "score": 65.5251,
          "outputTokens": 80473,
          "cost": 8.94966,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        }
      ]
    },
    {
      "id": "muse-spark",
      "name": "Muse Spark 1.1",
      "shortName": "Spark",
      "provider": "Meta",
      "color": "#72C7F4",
      "recommendedEffort": "XHigh",
      "readerLabel": "Historical anchor",
      "points": [
        {
          "effort": "XHigh",
          "score": 53.5422,
          "outputTokens": 34042,
          "cost": 1.428241,
          "evidence": "measured",
          "basis": "Measured Artificial Analysis Coding Agent Index v1.3 run, accessed July 31, 2026. Tested in the Opencode harness.",
          "sourceUrl": "https://artificialanalysis.ai/agents/coding-agents"
        }
      ]
    },
    {
      "id": "grok46",
      "name": "Grok 4.6",
      "shortName": "Grok 4.6",
      "provider": "xAI",
      "color": "#FF8A65",
      "recommendedEffort": "XHigh",
      "readerLabel": "New release",
      "points": [
        {
          "effort": "XHigh",
          "score": 70.8,
          "outputTokens": 41136,
          "cost": 2.81,
          "evidence": "provisional",
          "basis": "Provisional cross-suite marker. CursorBench 3.2 reports the score, average task cost, and total tokens for Grok 4.6 Extra High. These coordinates are internally consistent within CursorBench but are not directly comparable with Artificial Analysis Coding Agent Index circles.",
          "sourceUrl": "https://cursor.com/evals",
          "scoreMetric": "CursorBench 3.2 score",
          "costBasis": "CursorBench 3.2 average task cost",
          "tokenMetric": "CursorBench total tokens per task"
        }
      ]
    },
    {
      "id": "gemini37",
      "name": "Gemini 3.7 Flash",
      "shortName": "Gemini 3.7",
      "provider": "Google",
      "color": "#8AB4F8",
      "recommendedEffort": "High",
      "readerLabel": "New release",
      "points": [
        {
          "effort": "High",
          "score": 61.6,
          "outputTokens": 38448,
          "cost": 1.2,
          "evidence": "provisional",
          "basis": "Provisional cross-suite marker. CursorBench 3.2 reports the score, average task cost, and total tokens for Gemini 3.7 Flash High. These coordinates are internally consistent within CursorBench but are not directly comparable with Artificial Analysis Coding Agent Index circles.",
          "sourceUrl": "https://cursor.com/evals",
          "scoreMetric": "CursorBench 3.2 score",
          "costBasis": "CursorBench 3.2 average task cost",
          "tokenMetric": "CursorBench total tokens per task"
        }
      ]
    },
    {
      "id": "deepseek-v4-pro-0813",
      "name": "DeepSeek V4 Pro 0813",
      "shortName": "DS V4 Pro",
      "provider": "DeepSeek",
      "color": "#6DE0A8",
      "recommendedEffort": "Max",
      "readerLabel": "New release",
      "points": [
        {
          "effort": "Max",
          "score": 62.7,
          "outputTokens": 46300,
          "cost": 0.0839,
          "evidence": "provisional",
          "basis": "Provisional mixed-basis marker. DeepSeek reports a 62.7 DeepSWE score for V4 Pro 0813. The 46.3k token footprint reuses Artificial Analysis's V4 Flash 0731 general-task profile, and the $0.084 cost scales Flash's $0.027 task cost by the current 3.107x Pro-to-Flash token-price ratio. This is a directional pricing proxy, not measured V4 Pro task economics.",
          "sourceUrl": "https://api-docs.deepseek.com/updates#deepseek-v4-pro-update",
          "scoreMetric": "DeepSeek vendor DeepSWE",
          "costBasis": "Pricing-scaled V4 Flash general-task proxy",
          "tokenMetric": "V4 Flash general-task output-token proxy"
        }
      ]
    },
    {
      "id": "deepseek-v4-flash-0731",
      "name": "DeepSeek V4 Flash 0731",
      "shortName": "DS V4 Flash",
      "provider": "DeepSeek",
      "color": "#33C47A",
      "recommendedEffort": "Max",
      "readerLabel": "New release",
      "points": [
        {
          "effort": "Max",
          "score": 54.4,
          "outputTokens": 46300,
          "cost": 0.027,
          "evidence": "provisional",
          "basis": "Provisional mixed-basis marker. DeepSeek reports a 54.4 DeepSWE score for V4 Flash 0731; Artificial Analysis reports the $0.027 weighted general-task cost and 46.3k output-token footprint. The score and economics do not come from one common coding harness.",
          "sourceUrl": "https://artificialanalysis.ai/models/deepseek-v4-flash",
          "scoreMetric": "DeepSeek vendor DeepSWE",
          "costBasis": "Artificial Analysis Intelligence Index task suite",
          "tokenMetric": "AA Intelligence Index output tokens per task"
        }
      ]
    },
    {
      "id": "muse-spark-12",
      "name": "Muse Spark 1.2",
      "shortName": "Spark 1.2",
      "provider": "Meta",
      "color": "#72C7F4",
      "recommendedEffort": "XHigh",
      "readerLabel": "New release",
      "points": [
        {
          "effort": "XHigh",
          "score": 59.3,
          "outputTokens": 34042,
          "cost": 1.428241,
          "evidence": "provisional",
          "basis": "Provisional family-proxy marker. Meta reports a 59.3 DeepSWE 1.1 result for Muse Spark 1.2 with Muse Code. Cost and tokens reuse the measured Muse Spark 1.1 Artificial Analysis coding-agent point until independent 1.2 task economics are available.",
          "sourceUrl": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
          "scoreMetric": "Meta vendor DeepSWE 1.1",
          "costBasis": "Muse Spark 1.1 measured coding-agent family proxy",
          "tokenMetric": "Muse Spark 1.1 output-token family proxy"
        }
      ]
    }
  ],
  "cursorBench": {
    "version": "3.2",
    "methodology": "Selected operating points from CursorBench 3.2. Score, cost, total tokens, and steps are retained as a separate benchmark lens. Current Grok 4.6 and Gemini 3.7 results also appear as provisional diamonds in the opening discovery plot and must not be treated as Artificial Analysis Coding Agent Index measurements.",
    "points": [
      {
        "id": "grok46-xhigh",
        "model": "Grok 4.6",
        "effort": "XHigh",
        "provider": "xAI",
        "score": 70.8,
        "cost": 2.81,
        "totalTokens": 41136,
        "steps": 46,
        "color": "#FF8A65"
      },
      {
        "id": "fable5-max",
        "model": "Claude Fable 5",
        "effort": "Max",
        "provider": "Anthropic",
        "score": 70.5,
        "cost": 17.32,
        "totalTokens": 103525,
        "steps": 72,
        "color": "#B99AF8"
      },
      {
        "id": "opus5-max",
        "model": "Claude Opus 5",
        "effort": "Max",
        "provider": "Anthropic",
        "score": 70,
        "cost": 8.23,
        "totalTokens": 61838,
        "steps": 78,
        "color": "#F08A68"
      },
      {
        "id": "sol-max",
        "model": "GPT-5.6 Sol",
        "effort": "Max",
        "provider": "OpenAI",
        "score": 67.2,
        "cost": 5.69,
        "totalTokens": 28320,
        "steps": 48,
        "color": "#E3B341"
      },
      {
        "id": "grok45-high",
        "model": "Grok 4.5",
        "effort": "High",
        "provider": "xAI",
        "score": 66.7,
        "cost": 1.51,
        "totalTokens": 19521,
        "steps": 33,
        "color": "#EF7C5A",
        "caveat": "Historical anchor. Cursor reports a possible training-data advantage from an earlier Cursor codebase snapshot.",
        "visibility": "history"
      },
      {
        "id": "terra-max",
        "model": "GPT-5.6 Terra",
        "effort": "Max",
        "provider": "OpenAI",
        "score": 64.9,
        "cost": 2.31,
        "totalTokens": 32969,
        "steps": 47,
        "color": "#4FB6C2"
      },
      {
        "id": "luna-max",
        "model": "GPT-5.6 Luna",
        "effort": "Max",
        "provider": "OpenAI",
        "score": 61.1,
        "cost": 0.39,
        "totalTokens": 87973,
        "steps": 61,
        "color": "#42B883"
      },
      {
        "id": "gemini37-high",
        "model": "Gemini 3.7 Flash",
        "effort": "High",
        "provider": "Google",
        "score": 61.6,
        "cost": 1.2,
        "totalTokens": 38448,
        "steps": 99,
        "color": "#8AB4F8"
      },
      {
        "id": "kimi-k3-max",
        "model": "Kimi K3",
        "effort": "Max",
        "provider": "Moonshot AI",
        "score": 60.8,
        "cost": 2.7,
        "totalTokens": 38428,
        "steps": 57,
        "color": "#67D7E8"
      },
      {
        "id": "gpt55-high",
        "model": "GPT-5.5",
        "effort": "High",
        "provider": "OpenAI",
        "score": 58.4,
        "cost": 2.05,
        "totalTokens": 12183,
        "steps": 28,
        "color": "#CBD5E1",
        "visibility": "history"
      },
      {
        "id": "composer25",
        "model": "Composer 2.5",
        "effort": "Default",
        "provider": "Cursor",
        "score": 56.1,
        "cost": 0.44,
        "totalTokens": 14286,
        "steps": 33,
        "color": "#F2C14E"
      },
      {
        "id": "glm52-max",
        "model": "GLM-5.2",
        "effort": "Max",
        "provider": "Z.ai",
        "score": 55,
        "cost": 1.76,
        "totalTokens": 35946,
        "steps": 58,
        "color": "#5AA7FF"
      }
    ]
  },
  "releaseEvidence": [
    {
      "id": "deepseek-v4-pro-0813",
      "name": "DeepSeek V4 Pro 0813",
      "provider": "DeepSeek",
      "status": "Provisional master-plot diamond, common-harness economics pending",
      "summary": "DeepSeek V4 Pro 0813 is visible in the default plot using its vendor DeepSWE score and a pricing-scaled V4 Flash task-profile proxy. The diamond is directional, not a same-harness frontier point.",
      "metrics": [
        {
          "value": "62.7",
          "label": "Vendor DeepSWE"
        },
        {
          "value": "~$0.084",
          "label": "Pricing-scaled task proxy"
        },
        {
          "value": "~46.3k",
          "label": "Flash token-profile proxy"
        }
      ],
      "sourceUrl": "https://api-docs.deepseek.com/updates#deepseek-v4-pro-update",
      "secondarySourceUrl": "https://api-docs.deepseek.com/quick_start/pricing/",
      "secondarySourceLabel": "Review current pricing",
      "comparableToCodingPlot": false
    },
    {
      "id": "gemini-37-flash",
      "name": "Gemini 3.7 Flash",
      "provider": "Google",
      "status": "Provisional master-plot diamond from CursorBench 3.2",
      "summary": "Google lists Gemini 3.7 Flash as generally available. Its default diamond uses the complete CursorBench score, task cost, and total-token result; it remains cross-suite rather than Artificial Analysis frontier evidence.",
      "metrics": [
        {
          "value": "61.6%",
          "label": "CursorBench high"
        },
        {
          "value": "$1.20",
          "label": "Cost per task"
        },
        {
          "value": "38.4k",
          "label": "Total tokens"
        }
      ],
      "sourceUrl": "https://ai.google.dev/gemini-api/docs/changelog",
      "secondarySourceUrl": "https://cursor.com/evals",
      "secondarySourceLabel": "Review CursorBench",
      "comparableToCodingPlot": false
    },
    {
      "id": "grok-46",
      "name": "Grok 4.6",
      "provider": "xAI",
      "status": "Provisional master-plot diamond from CursorBench 3.2",
      "summary": "xAI confirms Grok 4.6 API availability. Its default diamond uses the complete CursorBench score, task cost, and total-token result, while Grok 4.5 remains a historical Artificial Analysis anchor.",
      "metrics": [
        {
          "value": "70.8%",
          "label": "CursorBench xhigh"
        },
        {
          "value": "$2.81",
          "label": "Cost per task"
        },
        {
          "value": "41.1k",
          "label": "Total tokens"
        }
      ],
      "sourceUrl": "https://docs.x.ai/developers/release-notes",
      "secondarySourceUrl": "https://cursor.com/evals",
      "secondarySourceLabel": "Review CursorBench",
      "comparableToCodingPlot": false
    },
    {
      "id": "deepseek-v4-flash-0731",
      "name": "DeepSeek V4 Flash 0731",
      "provider": "DeepSeek",
      "status": "Provisional master-plot diamond with mixed evidence",
      "summary": "The default diamond combines DeepSeek's vendor DeepSWE score with Artificial Analysis general-task cost and output tokens. Its low-cost position is useful, but it is not a same-harness coding-economics point.",
      "metrics": [
        {
          "value": "54.4",
          "label": "Vendor DeepSWE"
        },
        {
          "value": "$0.027",
          "label": "General task cost"
        },
        {
          "value": "46.3k",
          "label": "Output tokens per task"
        }
      ],
      "sourceUrl": "https://artificialanalysis.ai/models/deepseek-v4-flash",
      "comparableToCodingPlot": false
    },
    {
      "id": "muse-spark-12",
      "name": "Muse Spark 1.2 with Muse Code",
      "provider": "Meta",
      "status": "Provisional master-plot diamond with a family proxy",
      "summary": "The default diamond combines Meta's Muse Spark 1.2 DeepSWE result with the measured Muse Spark 1.1 task-cost and token profile. It shows a directional family position while independent 1.2 economics remain pending.",
      "metrics": [
        {
          "value": "82.9%",
          "label": "Terminal-Bench 2.1"
        },
        {
          "value": "59.3%",
          "label": "DeepSWE 1.1"
        },
        {
          "value": "~$1.43",
          "label": "Spark 1.1 task-cost proxy"
        }
      ],
      "sourceUrl": "https://research.meta.ai/blog/introducing-muse-code-and-muse-spark-1-2",
      "comparableToCodingPlot": false
    },
    {
      "id": "qwen38-max-preview",
      "name": "Qwen3.8-Max-Preview",
      "provider": "Alibaba Cloud",
      "status": "Hosted preview, comparable coding economics pending",
      "summary": "Qwen Code confirms the preview is available through the Token Plan. A public common-harness coding score, task cost, and token footprint are still required before plotting it.",
      "metrics": [
        {
          "value": "Preview",
          "label": "Availability"
        },
        {
          "value": "Pending",
          "label": "Coding Agent Index"
        },
        {
          "value": "Pending",
          "label": "Task cost and tokens"
        }
      ],
      "sourceUrl": "https://qwenlm.github.io/qwen-code-docs/en/blog/updates/weekly-update-2026-07-23/",
      "comparableToCodingPlot": false
    }
  ]
}
