{
  "name": "Ground Truth ratings registry",
  "url": "https://groundtruth.health/ratings/",
  "description": "Structured, versioned credibility ratings of AI claims in health, each traced to primary sources. Bands rate the credibility of the claim as stated — never system performance — and never rank systems.",
  "license": "https://creativecommons.org/licenses/by/4.0/",
  "scale": {
    "1": "Misleading",
    "2": "Overstated",
    "3": "Needs context",
    "4": "Holds up with conditions",
    "5": "Holds up"
  },
  "ratings": [
    {
      "subject": "The circulating claim that the PinPoint AI blood test is 99% accurate at both detecting and ruling out gynaecological cancer",
      "claim": "The PinPoint AI blood test is 99% accurate at both detecting gynaecological cancers and ruling them out.",
      "claimant": "As carried by The Guardian (8 July 2026) and repeated across the coverage · Ground Truth's restatement of the accuracy proposition, not a verbatim quotation from any single source — the Guardian's exact sentence is quoted in the article",
      "type": "claim",
      "band": 2,
      "bandLabel": "Overstated",
      "rubricVersion": "1.0",
      "rated": "2026-07-25",
      "url": "https://groundtruth.health/nhs-cancer-blood-test-99-percent/#rating",
      "appearance": {
        "name": "The Guardian, “Thousands of women could be spared painful cancer exam by new NHS AI blood test” (8 July 2026)",
        "url": "https://www.theguardian.com/society/2026/jul/08/thousands-of-women-could-be-spared-painful-cancer-exam-by-new-nhs-ai-blood-test"
      },
      "sources": [
        {
          "name": "Neal M et al., “Real-World Validation of PinPoint Blood Tests in the NHS”, Mayo Clin Proc Digit Health 2026;4(3):100382 (CC BY 4.0)",
          "url": "https://doi.org/10.1016/j.mcpdig.2026.100382"
        },
        {
          "name": "The same paper, open access via PubMed Central",
          "url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC13325911/"
        },
        {
          "name": "Savage R et al., development and validation of the PinPoint algorithms, BMJ Open 2022;12:e053590",
          "url": "https://bmjopen.bmj.com/content/12/4/e053590"
        },
        {
          "name": "The Guardian, “Thousands of women could be spared painful cancer exam by new NHS AI blood test”",
          "url": "https://www.theguardian.com/society/2026/jul/08/thousands-of-women-could-be-spared-painful-cancer-exam-by-new-nhs-ai-blood-test"
        },
        {
          "name": "NHS Innovation Accelerator, “New blood test could spare thousands of women invasive gynaecological diagnostics”",
          "url": "https://nhsaccelerator.com/insights/new-blood-test-could-spare-thousands-of-women-invasive-gynaecological-diagnostics/"
        },
        {
          "name": "NHS England, Cancer Waiting Times statistics",
          "url": "https://www.england.nhs.uk/statistics/statistical-work-areas/cancer-waiting-times/"
        },
        {
          "name": "Ground Truth field guide: How to read a “99% accurate” diagnostic-AI claim",
          "url": "/how-to-read-a-99-percent-accurate-claim/"
        }
      ],
      "corrections": []
    },
    {
      "subject": "Whether small, phone-deployable medical LLMs are safe to confirm emergency drug orders where no clinician is reachable",
      "claim": "MedGemma models can be adapted to run on mobile hardware and are “small enough to run offline” for health AI — the basis for deploying phone-sized medical AI to frontline care in low-connectivity settings.",
      "claimant": "Google · Health AI Developer Foundations · 2025–2026 (echoed by the World Bank’s “Small AI” agenda)",
      "type": "benchmark",
      "band": 1,
      "bandLabel": "Misleading",
      "rubricVersion": "1.0",
      "rated": "2026-07-24",
      "url": "https://groundtruth.health/on-device-medical-ai-dosing-safety/#rating",
      "appearance": {
        "name": "Google, “MedGemma: our most capable open models for health AI development” (2025)",
        "url": "https://research.google/blog/medgemma-our-most-capable-open-models-for-health-ai-development/"
      },
      "sources": [
        {
          "name": "Google, “MedGemma” launch — ‘run on a single GPU … mobile hardware’",
          "url": "https://research.google/blog/medgemma-our-most-capable-open-models-for-health-ai-development/"
        },
        {
          "name": "Google, MedGemma model card — clinical-use disclaimer",
          "url": "https://developers.google.com/health-ai-developer-foundations/medgemma/model-card"
        },
        {
          "name": "World Bank, “Small AI, Big Impact”",
          "url": "https://www.worldbank.org/en/topic/digital/brief/small-ai-big-impact"
        },
        {
          "name": "WHO, Managing Complications in Pregnancy and Childbirth (2nd ed.) — magnesium sulfate regimen (Box S-4)",
          "url": "https://iris.who.int/handle/10665/255760"
        },
        {
          "name": "HELIX trial — cooling raised mortality in low-income settings (Lancet Global Health 2021)",
          "url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC8371331/"
        },
        {
          "name": "Ground Truth on-device dosing benchmark — code, probe set, and per-response outputs (MIT)",
          "url": "https://github.com/groundtruth-health/medgemma-benchmark"
        },
        {
          "name": "Ground Truth, “Google made Gemma ‘medical.’ We gave it the job.”",
          "url": "/medgemma-medical-badge"
        }
      ],
      "corrections": []
    },
    {
      "subject": "The circulating claim that OpenAI's GPT-5.6 outperformed physicians on health evaluations",
      "claim": "OpenAI's GPT-5.6 outperforms physician responses in health evaluations — 60.5 versus 43.7.",
      "claimant": "Crypto Briefing, 11 July 2026, syndicated by KuCoin and others · fusing OpenAI's system card with an earlier paper",
      "type": "claim",
      "band": 2,
      "bandLabel": "Overstated",
      "rubricVersion": "1.0",
      "rated": "2026-07-23",
      "url": "https://groundtruth.health/openai-gpt56-beats-physicians/#rating",
      "appearance": {
        "name": "Crypto Briefing, “OpenAI's GPT-5.6 outperforms physician responses in health evaluations” (11 July 2026)",
        "url": "https://cryptobriefing.com/openai-gpt-56-outperforms-physicians-health-evaluations/"
      },
      "sources": [
        {
          "name": "OpenAI, HealthBench Professional (arXiv:2604.27470)",
          "url": "https://arxiv.org/abs/2604.27470"
        },
        {
          "name": "OpenAI, GPT-5.6 Preview System Card",
          "url": "https://deploymentsafety.openai.com/gpt-5-6-preview"
        },
        {
          "name": "OpenAI, “Making ChatGPT better for clinicians”",
          "url": "https://openai.com/index/making-chatgpt-better-for-clinicians/"
        },
        {
          "name": "OpenAI, “Improving health intelligence in ChatGPT” (GPT-5.5 Instant)",
          "url": "https://openai.com/index/improving-health-intelligence-in-chatgpt/"
        },
        {
          "name": "HealthBench Professional dataset (MIT licence)",
          "url": "https://openaipublic.blob.core.windows.net/simple-evals/healthbench_professional/assets.zip"
        }
      ],
      "corrections": []
    },
    {
      "subject": "Google MedGemma vs the Gemma base it was built from and a newer general Gemma, on clinical decision support",
      "claim": "MedGemma: our most capable open models for health AI development.",
      "claimant": "Google · Health AI Developer Foundations · 2025",
      "type": "benchmark",
      "band": 3,
      "bandLabel": "Needs context",
      "rubricVersion": "1.0",
      "rated": "2026-07-17",
      "url": "https://groundtruth.health/medgemma-medical-badge/#rating",
      "appearance": {
        "name": "Google, MedGemma Technical Report (arXiv:2507.05201)",
        "url": "https://arxiv.org/abs/2507.05201"
      },
      "sources": [
        {
          "name": "Google, MedGemma Technical Report (arXiv:2507.05201)",
          "url": "https://arxiv.org/abs/2507.05201"
        },
        {
          "name": "Ground Truth MedGemma benchmark — code and per-item outputs (MIT)",
          "url": "https://github.com/groundtruth-health/medgemma-benchmark"
        }
      ],
      "corrections": []
    },
    {
      "subject": "Blind-sweep ultrasound AI — gestational-age estimation vs. expert fetal biometry",
      "claim": "Between 14 and 27 weeks’ gestation, novice users with no prior training in ultrasonography estimated GA as accurately … as credentialed sonographers performing standard biometry.",
      "claimant": "Stringer et al. · JAMA conclusion · Aug 2024",
      "type": "benchmark",
      "band": 4,
      "bandLabel": "Holds up with conditions",
      "rubricVersion": "1.0",
      "rated": "2026-07-14",
      "url": "https://groundtruth.health/ai-ultrasound-as-accurate-as-sonographers/#rating",
      "appearance": {
        "name": "Stringer et al., “Diagnostic Accuracy of an Integrated AI Tool to Estimate Gestational Age From Blind Ultrasound Sweeps” (JAMA, 2024)",
        "url": "https://jamanetwork.com/journals/jama/fullarticle/2821666"
      },
      "sources": [
        {
          "name": "Stringer et al., “Diagnostic Accuracy of an Integrated AI Tool to Estimate Gestational Age From Blind Ultrasound Sweeps” (JAMA, 2024)",
          "url": "https://jamanetwork.com/journals/jama/fullarticle/2821666"
        },
        {
          "name": "NCT05433519 statistical analysis plan",
          "url": "https://cdn.clinicaltrials.gov/large-docs/19/NCT05433519/SAP_001.pdf"
        }
      ],
      "corrections": []
    },
    {
      "subject": "Hippocratic AI — Polaris vs. human nurses (Polaris 1.0 evaluation)",
      "claim": "Polaris performs on par with human nurses on aggregate across dimensions such as medical safety, clinical readiness, patient education, conversational quality, and bedside manner.",
      "claimant": "Hippocratic AI · Mar 2024",
      "type": "benchmark",
      "band": 3,
      "bandLabel": "Needs context",
      "rubricVersion": "1.0",
      "rated": "2026-07-10",
      "url": "https://groundtruth.health/hippocratic-ai-polaris-on-par-with-nurses/#rating",
      "appearance": {
        "name": "Mukherjee et al. (Hippocratic AI), “Polaris: A Safety-focused LLM Constellation Architecture for Healthcare” (2024)",
        "url": "https://arxiv.org/abs/2403.13313"
      },
      "sources": [
        {
          "name": "Mukherjee et al. (Hippocratic AI), “Polaris: A Safety-focused LLM Constellation Architecture for Healthcare” (arXiv:2403.13313, 2024)",
          "url": "https://arxiv.org/abs/2403.13313"
        },
        {
          "name": "Hippocratic AI, “Polaris 5.0” launch (PR Newswire, 30 Apr 2026)",
          "url": "https://www.prnewswire.com/news-releases/hippocratic-ai-launches-polaris-5-0-the-first-evidence--based-ai-for-healthcare-proven-to-outperform-every-frontier-model-on-critical-medical-tasks-and-safety-302758998.html"
        },
        {
          "name": "Hippocratic AI × DigitalOcean, “…10 Million Patient Calls at 99.9% Clinical Safety” (27 May 2026)",
          "url": "https://investors.digitalocean.com/news/news-details/2026/Hippocratic-AI-Scales-to-10-Million-Patient-Calls-at-99-9-Clinical-Safety-on-DigitalOceans-AI-Native-Cloud-powered-by-NVIDIA-Blackwell-Ultra-GPUs/default.aspx"
        },
        {
          "name": "Hippocratic AI, RWE-LLM: Real-World Evaluation of LLMs in Healthcare (medRxiv 2025.03.17.25324157, preprint)",
          "url": "https://www.medrxiv.org/content/10.1101/2025.03.17.25324157v1"
        },
        {
          "name": "ClinicalTrials.gov — search for sponsor “Hippocratic AI” (no registered trial)",
          "url": "https://clinicaltrials.gov/search?term=Hippocratic%20AI"
        }
      ],
      "corrections": []
    },
    {
      "subject": "Microsoft AI — MAI-DxO on the Sequential Diagnosis Benchmark",
      "claim": "[The] Microsoft AI Diagnostic Orchestrator (MAI-DxO) correctly diagnoses up to 85% of NEJM case proceedings, a rate more than four times higher than a group of experienced physicians.",
      "claimant": "Microsoft AI · Jun 2025",
      "type": "benchmark",
      "band": 2,
      "bandLabel": "Overstated",
      "rubricVersion": "1.0",
      "rated": "2026-07-10",
      "url": "https://groundtruth.health/microsoft-mai-dxo-medical-superintelligence/#rating",
      "appearance": {
        "name": "Microsoft AI, “The Path to Medical Superintelligence” (2025)",
        "url": "https://microsoft.ai/news/the-path-to-medical-superintelligence/"
      },
      "sources": [
        {
          "name": "Nori et al. (Microsoft AI), “Sequential Diagnosis with Language Models” (arXiv:2506.22405, 2025)",
          "url": "https://arxiv.org/abs/2506.22405"
        },
        {
          "name": "Microsoft AI, “The Path to Medical Superintelligence” (2025)",
          "url": "https://microsoft.ai/news/the-path-to-medical-superintelligence/"
        },
        {
          "name": "Microsoft AI, “Towards Humanist Superintelligence” (2025)",
          "url": "https://microsoft.ai/news/towards-humanist-superintelligence/"
        }
      ],
      "corrections": []
    },
    {
      "subject": "OpenAI × Penda Health — AI Consult",
      "claim": "[Clinicians using the AI Consult copilot had] a 16% relative reduction in diagnostic errors and a 13% reduction in treatment errors compared to those without.",
      "claimant": "OpenAI & Penda Health · Jul 2025",
      "type": "deployment",
      "band": 3,
      "bandLabel": "Needs context",
      "rubricVersion": "1.0",
      "rated": "2026-07-09",
      "url": "https://groundtruth.health/openai-penda-health-medical-errors/#rating",
      "appearance": {
        "name": "OpenAI, “Pioneering an AI clinical copilot with Penda Health” (2025)",
        "url": "https://openai.com/index/ai-clinical-copilot-penda-health/"
      },
      "sources": [
        {
          "name": "Penda Health & OpenAI, “AI-based Clinical Decision Support for Primary Care: A Real-World Study” (arXiv:2507.16947, 2025)",
          "url": "https://arxiv.org/abs/2507.16947"
        },
        {
          "name": "OpenAI, “Pioneering an AI clinical copilot with Penda Health” (2025)",
          "url": "https://openai.com/index/ai-clinical-copilot-penda-health/"
        }
      ],
      "corrections": []
    }
  ]
}
