{
  "schemaVersion": 1,
  "retrievedAt": "2026-09-05T06:51:00.680Z",
  "license": "CC BY 4.0",
  "canonicalPage": "https://llm-stats.com/best-ai-for-x-ray-analysis",
  "methodologyUrl": "https://llm-stats.com/research/best-ai-for-x-ray-analysis/methodology.md",
  "scope": "Underlying model evidence from selected chest X-ray and radiology benchmarks; not a ranking of clinical products or medical devices",
  "aggregationRule": "No cross-benchmark average is calculated because tasks, samples, and metrics differ. Evidence coverage and per-benchmark scores are reported separately.",
  "selectedBenchmarkIds": [
    "chexpert-cxr",
    "mimic-cxr",
    "vqa-rad",
    "slakevqa"
  ],
  "modelCoverage": [
    {
      "model_id": "medgemma-4b-it",
      "model_name": "MedGemma 4B IT",
      "organization_name": "Google",
      "benchmarkCount": 4,
      "benchmarkIds": [
        "chexpert-cxr",
        "mimic-cxr",
        "vqa-rad",
        "slakevqa"
      ]
    },
    {
      "model_id": "qwen3.5-122b-a10b",
      "model_name": "Qwen3.5-122B-A10B",
      "organization_name": "Alibaba Cloud / Qwen Team",
      "benchmarkCount": 1,
      "benchmarkIds": [
        "slakevqa"
      ]
    },
    {
      "model_id": "qwen3.5-27b",
      "model_name": "Qwen3.5-27B",
      "organization_name": "Alibaba Cloud / Qwen Team",
      "benchmarkCount": 1,
      "benchmarkIds": [
        "slakevqa"
      ]
    },
    {
      "model_id": "qwen3.5-35b-a3b",
      "model_name": "Qwen3.5-35B-A3B",
      "organization_name": "Alibaba Cloud / Qwen Team",
      "benchmarkCount": 1,
      "benchmarkIds": [
        "slakevqa"
      ]
    }
  ],
  "limitations": [
    "Current comparable model coverage is sparse.",
    "VQA-RAD and SLAKE contain mixed radiology modalities and are not X-ray-only clinical reader studies.",
    "Benchmark scores do not establish diagnosis quality, device authorization, clinical utility, or improved patient outcomes.",
    "Model evidence does not represent the complete product, interface, workflow, privacy controls, or monitored deployment."
  ],
  "records": [
    {
      "benchmarkId": "chexpert-cxr",
      "benchmarkDescription": "CheXpert is a large dataset of 224,316 chest radiographs from 65,240 patients for automated chest X-ray interpretation. The dataset includes uncertainty labels for 14 medical observations extracted from radiology reports. It serves as a benchmark for developing and evaluating automated chest radiograph interpretation models.",
      "benchmarkModelCount": 1,
      "rankWithinBenchmark": 1,
      "modelId": "medgemma-4b-it",
      "model": "MedGemma 4B IT",
      "organization": "Google",
      "reportedScore": 0.481,
      "normalizedScore": 0.481,
      "sourceVerifiedInLlmStats": false
    },
    {
      "benchmarkId": "mimic-cxr",
      "benchmarkDescription": "MIMIC-CXR is a large publicly available dataset of chest radiographs with free-text radiology reports. Contains 377,110 images corresponding to 227,835 radiographic studies from 65,379 patients at Beth Israel Deaconess Medical Center. The dataset is de-identified and widely used for medical imaging research, automated report generation, and medical AI development.",
      "benchmarkModelCount": 1,
      "rankWithinBenchmark": 1,
      "modelId": "medgemma-4b-it",
      "model": "MedGemma 4B IT",
      "organization": "Google",
      "reportedScore": 0.889,
      "normalizedScore": 0.889,
      "sourceVerifiedInLlmStats": false
    },
    {
      "benchmarkId": "vqa-rad",
      "benchmarkDescription": "VQA-RAD (Visual Question Answering in Radiology) is the first manually constructed dataset of medical visual question answering containing 3,515 clinically generated visual questions and answers about radiology images. The dataset includes questions created by clinical trainees on 315 radiology images from MedPix covering head, chest, and abdominal scans, designed to support AI development for medical image analysis and improve patient care.",
      "benchmarkModelCount": 1,
      "rankWithinBenchmark": 1,
      "modelId": "medgemma-4b-it",
      "model": "MedGemma 4B IT",
      "organization": "Google",
      "reportedScore": 0.499,
      "normalizedScore": 0.499,
      "sourceVerifiedInLlmStats": false
    },
    {
      "benchmarkId": "slakevqa",
      "benchmarkDescription": "A semantically-labeled knowledge-enhanced dataset for medical visual question answering. Contains 642 radiology images (CT scans, MRI scans, X-rays) covering five body parts and 14,028 bilingual English-Chinese question-answer pairs annotated by experienced physicians. Features comprehensive semantic labels and a structural medical knowledge base with both vision-only and knowledge-based questions requiring external medical knowledge reasoning.",
      "benchmarkModelCount": 4,
      "rankWithinBenchmark": 1,
      "modelId": "qwen3.5-122b-a10b",
      "model": "Qwen3.5-122B-A10B",
      "organization": "Alibaba Cloud / Qwen Team",
      "reportedScore": 0.816,
      "normalizedScore": 0.816,
      "sourceVerifiedInLlmStats": false
    },
    {
      "benchmarkId": "slakevqa",
      "benchmarkDescription": "A semantically-labeled knowledge-enhanced dataset for medical visual question answering. Contains 642 radiology images (CT scans, MRI scans, X-rays) covering five body parts and 14,028 bilingual English-Chinese question-answer pairs annotated by experienced physicians. Features comprehensive semantic labels and a structural medical knowledge base with both vision-only and knowledge-based questions requiring external medical knowledge reasoning.",
      "benchmarkModelCount": 4,
      "rankWithinBenchmark": 2,
      "modelId": "qwen3.5-27b",
      "model": "Qwen3.5-27B",
      "organization": "Alibaba Cloud / Qwen Team",
      "reportedScore": 0.8,
      "normalizedScore": 0.8,
      "sourceVerifiedInLlmStats": false
    },
    {
      "benchmarkId": "slakevqa",
      "benchmarkDescription": "A semantically-labeled knowledge-enhanced dataset for medical visual question answering. Contains 642 radiology images (CT scans, MRI scans, X-rays) covering five body parts and 14,028 bilingual English-Chinese question-answer pairs annotated by experienced physicians. Features comprehensive semantic labels and a structural medical knowledge base with both vision-only and knowledge-based questions requiring external medical knowledge reasoning.",
      "benchmarkModelCount": 4,
      "rankWithinBenchmark": 3,
      "modelId": "qwen3.5-35b-a3b",
      "model": "Qwen3.5-35B-A3B",
      "organization": "Alibaba Cloud / Qwen Team",
      "reportedScore": 0.787,
      "normalizedScore": 0.787,
      "sourceVerifiedInLlmStats": false
    },
    {
      "benchmarkId": "slakevqa",
      "benchmarkDescription": "A semantically-labeled knowledge-enhanced dataset for medical visual question answering. Contains 642 radiology images (CT scans, MRI scans, X-rays) covering five body parts and 14,028 bilingual English-Chinese question-answer pairs annotated by experienced physicians. Features comprehensive semantic labels and a structural medical knowledge base with both vision-only and knowledge-based questions requiring external medical knowledge reasoning.",
      "benchmarkModelCount": 4,
      "rankWithinBenchmark": 4,
      "modelId": "medgemma-4b-it",
      "model": "MedGemma 4B IT",
      "organization": "Google",
      "reportedScore": 0.623,
      "normalizedScore": 0.623,
      "sourceVerifiedInLlmStats": false
    }
  ]
}
