[
  {
    "metric": "DeepSWE",
    "version": "v1.1",
    "score": 61.7,
    "unit": "%",
    "category": "Coding",
    "modelVersion": "Mistral Large 4 public preview",
    "sourceUrl": "https://mistral.ai/news/mistral-large-4/",
    "sourceDate": "2026-10-06",
    "checkedAt": "2026-10-07",
    "evidenceType": "Vendor-reported",
    "attribution": "Artificial Analysis",
    "notes": "Agent harness, inference budget and repeated-run uncertainty not independently checked."
  },
  {
    "metric": "SWE-Atlas-QnA",
    "version": "Not specified in release",
    "score": 59.4,
    "unit": "%",
    "category": "Coding",
    "modelVersion": "Mistral Large 4 public preview",
    "sourceUrl": "https://mistral.ai/news/mistral-large-4/",
    "sourceDate": "2026-10-06",
    "checkedAt": "2026-10-07",
    "evidenceType": "Vendor-reported",
    "attribution": "Artificial Analysis",
    "notes": "Dataset revision and evaluation configuration not independently checked."
  },
  {
    "metric": "Terminal-Bench",
    "version": "4.0",
    "score": 28.3,
    "unit": "%",
    "category": "Terminal tasks",
    "modelVersion": "Mistral Large 4 public preview",
    "sourceUrl": "https://mistral.ai/news/mistral-large-4/",
    "sourceDate": "2026-10-06",
    "checkedAt": "2026-10-07",
    "evidenceType": "Vendor-reported",
    "attribution": "Artificial Analysis",
    "notes": "Agent harness, tool permissions and inference budget not independently checked."
  },
  {
    "metric": "Coding Agent Index",
    "version": "Not specified in release",
    "score": 49.8,
    "unit": "%",
    "category": "Composite index",
    "modelVersion": "Mistral Large 4 public preview",
    "sourceUrl": "https://mistral.ai/news/mistral-large-4/",
    "sourceDate": "2026-10-06",
    "checkedAt": "2026-10-07",
    "evidenceType": "Vendor-reported",
    "attribution": "Artificial Analysis",
    "notes": "Composite result, not a separate task pass rate. Weighting not independently checked."
  }
]