<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e82360</article-id><article-id pub-id-type="doi">10.2196/82360</article-id><article-categories><subj-group subj-group-type="heading"><subject>Viewpoint</subject></subj-group></article-categories><title-group><article-title>From Data Poverty to Data Sovereignty: Operationalizing Gold-Standard Biomedical Datasets in Low- and Middle-Income Countries</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Rana</surname><given-names>Shweta</given-names></name><degrees>MPH</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Bhagat</surname><given-names>Pranshu</given-names></name><degrees>BTech</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Pal</surname><given-names>Debnath</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Pati</surname><given-names>Sanghamitra</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Singh</surname><given-names>Harpreet</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff6">6</xref></contrib></contrib-group><aff id="aff1"><institution>Division of Development Research, Indian Council of Medical Research</institution><addr-line>Ansari Nagar</addr-line><addr-line>Delhi</addr-line><addr-line>New Delhi</addr-line><country>India</country></aff><aff id="aff2"><institution>Corporate Strategy &#x0026; Quality (CEO's office), Piramal Group &#x0026; Strategy</institution><addr-line>Mumbai</addr-line><addr-line>Maharashtra</addr-line><country>India</country></aff><aff id="aff3"><institution>Wadhwani Institute of Artificial Intelligence</institution><addr-line>New Delhi</addr-line><addr-line>New Delhi</addr-line><country>India</country></aff><aff id="aff4"><institution>Department of Computational and Data Sciences, Indian Institute of Science (IISc)</institution><addr-line>Bangalore</addr-line><addr-line>Karnataka</addr-line><country>India</country></aff><aff id="aff5"><institution>Indian Council of Medical Research</institution><addr-line>Delhi</addr-line><addr-line>New Delhi</addr-line><country>India</country></aff><aff id="aff6"><institution>Faculty of Medical Sciences, Academy of Scientific and Innovative Research</institution><addr-line>Ghaziabad</addr-line><addr-line>Uttar Pradesh</addr-line><country>India</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Benis</surname><given-names>Arriel</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Zhang</surname><given-names>Jun</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Kulasegaram</surname><given-names>Kulamakan</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Harpreet Singh, PhD, Division of Development Research, Indian Council of Medical Research, Ansari Nagar, Delhi, New Delhi, India, 91 9999496965; <email>hsingh@bmi.icmr.org.in</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>10</day><month>8</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e82360</elocation-id><history><date date-type="received"><day>13</day><month>08</month><year>2025</year></date><date date-type="rev-recd"><day>23</day><month>02</month><year>2026</year></date><date date-type="accepted"><day>06</day><month>03</month><year>2026</year></date></history><copyright-statement>&#x00A9; Shweta Rana, Pranshu Bhagat, Debnath Pal, Sanghamitra Pati, Harpreet Singh. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 10.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e82360"/><abstract><p>AI is rapidly expanding in health care. However, there is a significant underrepresentation of low- and middle-income countries (LMICs) in datasets used to train AI applications. Current Food and Drug Administration (FDA)&#x2013;approved AI tools predominantly use data from high-income countries, with fewer than 4% reporting geographic or racial diversity. This geographic skew leads to significant performance degradation when these tools are used in LMIC populations, perpetuating an equity crisis where health burdens are highest.</p><p>Addressing this disparity, we highlight the Medical Imaging Datasets for India (MIDAS) initiative as a viable model to transition LMICs from &#x201C;data poverty&#x201D; to &#x201C;data sovereignty.&#x201D; MIDAS uses a rigorous, 4-domain Dataset Quality Matrix to ensure representativeness, documentation, technical fidelity, and governance, thereby creating openly benchmarked, gold-standard datasets tailored to local contexts. Initial releases, including datasets for oral and dural lesions, demonstrate the feasibility and practical value of developing robust, generalizable AI models.</p><p>Furthermore, we propose a multilateral South-South Data Commons structured around 3 foundational pillars: a harmonized dataset-grading rubric, distributed custodial governance, and outcome-linked incentives. This infrastructure supports local stewardship, encourages global collaboration, and ensures that quality benchmarks drive financial incentives for dataset expansion and diversity.</p><p>This proposed framework not only positions LMICs as autonomous data stewards but also enhances global AI equity. By institutionalizing quality control, interoperability, and outcome accountability, LMICs can transform from passive data consumers into active contributors to essential, trustworthy, and globally relevant biomedical datasets.</p></abstract><kwd-group><kwd>data poverty</kwd><kwd>data sovereignty</kwd><kwd>gold standard</kwd><kwd>low- and middle-income country</kwd><kwd>LMIC</kwd><kwd>biomedical datasets</kwd></kwd-group></article-meta></front><body><sec id="s1"><title>Background</title><p>The last decade has seen an explosion of AI tools approved for clinical use, yet fewer than 4% of the studies submitted to the US Food and Drug Administration (FDA) reported any racial or geographic diversity, and almost none disclosed socioeconomic data [<xref ref-type="bibr" rid="ref1">1</xref>]. Systematic assessments show that the publicly accessible training corpora for clinical AI are drawn disproportionately from a small cluster of wealthy nations, leaving most of the world essentially invisible in the data landscape. Among 7314 PubMed-indexed clinical AI studies analyzed in a 2022 global review, 71% of the datasets originated in just 10 high-income countries (HICs), with the United States alone contributing 41%. This underscores the pronounced geographic skew that shapes model development [<xref ref-type="bibr" rid="ref2">2</xref>]. Predictably, the devices trained on data from HIC cohorts have shown unstable performance when deployed in low- and middle-income country (LMIC) settings, undermining confidence in AI as an enabler of universal health coverage [<xref ref-type="bibr" rid="ref3">3</xref>]. The resulting &#x201C;health-data poverty&#x201D; [<xref ref-type="bibr" rid="ref4">4</xref>] is no longer merely a research inconvenience; it is an equity crisis that threatens to widen outcome gaps precisely where disease burdens are greatest [<xref ref-type="bibr" rid="ref5">5</xref>].</p><p>In this paper, we outline how LMIC-led, quality-graded, openly benchmarked data frameworks, exemplified by the Medical Imaging Datasets for India (MIDAS) program, can provide a practical solution for transitioning from data poverty to data sovereignty. Data sovereignty refers to institutional decision-making authority and accountability over the full data lifecycle (including collection, curation, access terms, downstream validation, and benefit-sharing) exercised by the institutions closest to the data subjects. It is distinct from data localization, which is primarily a jurisdictional requirement governing where data are stored or processed, and from data access, which concerns permissions to use data without necessarily shifting custodianship or governance authority. By enabling locally governed, demographically annotated datasets, this approach also supports equity for historically marginalized populations within LMICs, including rural communities, ethnic minority groups, and socioeconomically disadvantaged groups whose clinical representation is systematically absent from global AI benchmarks. We also propose a multilateral South-South Data Commons built on 3 pillars: a harmonized dataset-grading rubric, distributed custodial governance, and outcome-linked incentives for AI-enabled products.</p></sec><sec id="s2"><title>A Landscape Defined by Opaqueness and Asymmetry</title><p>Commercial enthusiasm has yielded more than 1000 FDA-listed AI-enabled devices, yet transparency gaps remain striking. Only 3.6% of regulatory dossiers disclose race or ethnicity, and fewer than 1% report socioeconomic indicators. Among the regulatory dossiers, 46.1% provided detailed performance data, 1.9% provided a link to a scientific publication, and only 9% included a prospective study for postmarket surveillance [<xref ref-type="bibr" rid="ref1">1</xref>]. Similar deficits plague public benchmark datasets: a systematic review in <italic>The Lancet Digital Health</italic> identified major documentation shortfalls in more than 75% of the COVID-19 pandemic datasets most frequently reused for AI development [<xref ref-type="bibr" rid="ref6">6</xref>].</p><p>Although radiology societies advocate for radically open corpora, the public availability of population-level imaging datasets remains scarce, and nearly all such datasets originate from high-income settings. Many studies have highlighted this geographical skew in datasets. Using the MEDLINE database, Google&#x2019;s search engine, and Google Dataset Search, a set of 94 open access datasets containing 507,724 images and 125 videos from 122,364 patients were identified. Most datasets originated from Asia, North America, and Europe [<xref ref-type="bibr" rid="ref7">7</xref>]. A 2023 review of 110 magnetic resonance imaging (MRI) datasets found no datasets from LMICs, underscoring how this geographic skew hampers the development of robust, generalizable AI models for global use [<xref ref-type="bibr" rid="ref8">8</xref>]. The resulting geographic skew is now well cataloged: models trained on US and EU chest radiograph data lose up to 25% points in the area under the curve when validated in African populations, even after standard transfer learning [<xref ref-type="bibr" rid="ref9">9</xref>]. The WHO 2024 guidance on AI governance warns that nonrepresentative datasets can entrench structural inequities at scale and urges member states to cultivate AI learning ecosystems that advance equity and safety [<xref ref-type="bibr" rid="ref10">10</xref>].</p></sec><sec id="s3"><title>Data Sovereignty: Beyond Access to Agency</title><p>The current solutions for providing LMICs with access to datasets from HICs treat the problem as a simple data shortage rather than as an issue of unequal power. The concept of data sovereignty reframes the debate: the origin of the data determines both their scientific reliability and the authority over their use. A <italic>BMJ Global Health</italic> scoping review shows that local stewardship improves adherence to context-specific ethical norms and accelerates regulatory approvals [<xref ref-type="bibr" rid="ref11">11</xref>]. Sovereignty, however, demands more than national firewalls; it requires interoperable standards that enable cross-border collaboration without ceding custodial control. In this manuscript, data sovereignty is not treated as synonymous with data localization or access restriction, but as institutional agency across the AI development, validation, and deployment lifecycle. <xref ref-type="table" rid="table1">Table 1</xref> summarizes the differences between the prevailing HIC-centric AI development pipeline and the proposed LMIC data sovereignty model across key stages of the AI lifecycle. One concrete step in this direction is the MIDAS platform, which is a collaborative effort between the Indian Council of Medical Research (ICMR), the Indian Institute of Science (IISc), and AI and Robotics Technology Park (ARTPARK) to develop high-quality, standardized, and diverse datasets for developing contextually relevant, AI-driven health care solutions in India. Using a hub-and-spoke model, MIDAS enables the systematic collection, annotation, and harmonization of multimodal clinical data.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Comparison of the current high-income country (HIC)&#x2013;centric AI development pipeline and the proposed low- and middle-income country (LMIC) data sovereignty model.<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Steps in the AI pipeline</td><td align="left" valign="bottom">Current HIC-centric model</td><td align="left" valign="bottom">Proposed LMIC data sovereignty model</td></tr></thead><tbody><tr><td align="left" valign="top">Dataset origin</td><td align="left" valign="top">Data sourced from HICs (eg, United States and Europe)</td><td align="left" valign="top">Data sourced from local LMIC institutions through MIDAS<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> or similar frameworks</td></tr><tr><td align="left" valign="top">Data quality oversight</td><td align="left" valign="top">Minimal or nontransparent grading; dataset quality varies</td><td align="left" valign="top">Dataset quality graded using structured frameworks (eg, MIDAS Quality Matrix)</td></tr><tr><td align="left" valign="top">Demographic representation</td><td align="left" valign="top">Often lacks racial, geographic, and socioeconomic diversity</td><td align="left" valign="top">Designed to capture local population diversity (age, gender, ethnicity, and geography)</td></tr><tr><td align="left" valign="top">Model development</td><td align="left" valign="top">Trained on HIC datasets, often generalized for global use</td><td align="left" valign="top">Trained and validated on representative LMIC datasets</td></tr><tr><td align="left" valign="top">Evaluation and audit</td><td align="left" valign="top">Performance metrics reported by developers, often lacking external validation</td><td align="left" valign="top">External, privacy-preserving audits with third-party evaluators</td></tr><tr><td align="left" valign="top">Regulatory pathway</td><td align="left" valign="top">US Food and Drug Administration approval or <italic>Conformit&#x00E9; Europ&#x00E9;enne</italic> marking based on HIC data; limited LMIC relevance</td><td align="left" valign="top">Contextualized validation mechanisms with LMIC participation</td></tr><tr><td align="left" valign="top">Deployment and use</td><td align="left" valign="top">Deployed globally without local adaptation; performance issues common in LMICs</td><td align="left" valign="top">Deployed with outcome-linked incentives tied to local performance and equity metrics</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>This table contrasts the prevailing HIC-driven AI development pathway with a proposed LMIC-led framework grounded in data sovereignty principles. The proposed framework emphasizes quality grading (eg, Medical Imaging Datasets for India), local custodianship, demographic diversity, and incentivized deployment through validated benchmarks.</p></fn><fn id="table1fn2"><p><sup>b</sup>MIDAS: Medical Imaging Datasets for India.</p></fn></table-wrap-foot></table-wrap><p>To operationalize these principles, MIDAS uses a 4-domain Dataset Quality Matrix covering representativeness, documentation, technical fidelity, and governance. Scores (0&#x2010;100) map onto Bronze, Silver, Gold, Platinum, and Diamond grades, with an external audit conducted prior to public release (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The first gold-graded corpus, comprising 1000 oral-lesion images drawn from 6 Indian cancer centers, was released in October 2024 under a Creative Commons Attribution&#x2013;NonCommercial (CC-BY-NC) license [<xref ref-type="bibr" rid="ref12">12</xref>]. Recently, the dataset was onboarded onto AIKosh [<xref ref-type="bibr" rid="ref13">13</xref>], India&#x2019;s national AI repository. Within 1 month, the dataset ranked third among trending datasets on the platform (out of 5577 datasets) and was downloaded 738 times, indicating early uptake by the AI research and developer community. A Gold-graded dural lesion dataset followed in March 2025, codeveloped with the Department of Neurosurgery, All India Institute of Medical Sciences (AIIMS), New Delhi. Crucially, both datasets are anonymized; however, they include demographic fields such as age, gender, and ethnicity, which are required for developing robust AI applications, as well as versioned consent artifacts, thereby addressing the transparency gaps currently evident in FDA filings.</p><p>Global data-sovereignty efforts such as the African Health Data Space and the Organisation for Economic Co-operation and Development (OECD) Health Data Governance Principles have emphasized legal interoperability, ethical safeguards, and cross-border data-sharing norms. These frameworks are critical for establishing trust and harmonization but largely stop short of operationalizing sovereignty within the AI development lifecycle itself. MIDAS complements these initiatives by introducing a quantitative, auditable dataset-grading framework that links local stewardship to model validation, benchmarking, and outcome-linked incentives. Rather than focusing solely on access governance or data flows, MIDAS embeds sovereignty at the level of dataset readiness and evaluative authority, enabling LMIC institutions to influence how AI systems are tested, certified, and rewarded. This positions MIDAS not as an alternative to existing governance frameworks, but as an implementation layer that translates global principles into measurable, AI-relevant practice.</p></sec><sec id="s4"><title>Toward a South-South Data Commons</title><p>Building on the MIDAS foundation, we propose a 3-pillar framework for establishing a South-South Data Commons that transcends bilateral collaborations to create a genuinely multilateral infrastructure (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Three pillars of the South-South Data Commons Framework. A self-reinforcing cycle representing how low- and middle-income countries (LMICs) can transition from data poverty to data sovereignty. Local institutions generate graded datasets (eg, via Medical Imaging Datasets for India), which are used by AI developers and externally validated. Models that meet performance thresholds receive outcome-linked incentives, supporting reinvestment in further data curation and strengthening the ecosystem.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e82360_fig01.png"/></fig></sec><sec id="s5"><title>Harmonized Grading Rubric</title><p>Replicating MIDAS without coordination risks a balkanized landscape of incompatible locally developed metrics. We therefore endorse the adoption of a common, open-source rubric extending MIDAS to nonimaging modalities and aligning it with the emerging Medical AI Data for All (MAIDA) framework for global medical datasets [<xref ref-type="bibr" rid="ref14">14</xref>]. These perspectives have called for a &#x201C;shared quality lingua franca&#x201D; to facilitate cross-registry discovery [<xref ref-type="bibr" rid="ref15">15</xref>]; a South-South rubric jointly maintained by the ICMR, the IISc, and the African Health Data Space could satisfy that mandate while preserving local autonomy [<xref ref-type="bibr" rid="ref16">16</xref>].</p></sec><sec id="s6"><title>Distributed Custodial Governance</title><p>Distributed custodial governance within the proposed South-South Data Commons is operationalized through a clear separation of roles across data stewardship, technical validation, and trust certification. Primary custodianship remains with the institutions closest to the data subjects, which retain authority over consent management, ethics approvals, and long-term stewardship of datasets. Technical validation can be distributed and conducted by designated nodal evaluators such as public research institutions or accredited AI evaluation laboratories with domain expertise using predefined grading rubrics such as the MIDAS Quality Matrix. Trust and certification are established through publicly accessible, versioned artifacts, including dataset grades, audit summaries, and metadata records hosted on national platforms such as AIKosh. The Trustworthy Evaluation of Clinical AI consortium has demonstrated that external evaluators can audit diabetic retinopathy algorithms using encrypted uploads without exposing patient-level data [<xref ref-type="bibr" rid="ref17">17</xref>]. This functional separation allows local institutions to retain sovereignty while enabling credible, repeatable, and scalable validation, thereby preventing centralization and reinforcing mutual accountability within the Commons [<xref ref-type="bibr" rid="ref17">17</xref>].</p></sec><sec id="s7"><title>Outcome-Linked Procurement Incentives</title><p>Market pull is indispensable. England&#x2019;s National Health Service (NHS) AI Award releases final-stage funds only when applicants&#x2019; algorithms meet accuracy and fairness targets on the test partition of the National COVID-19 Chest Imaging Database [<xref ref-type="bibr" rid="ref18">18</xref>]. By hard-wiring gold-standard dataset performance into payment schedules, these programs turn data quality into a revenue driver and create a self-reinforcing loop: vendors eager for outcome bonuses invest in expanding and diversifying the very datasets that will be used to evaluate the next procurement round. Governance is operationalized through transparent evaluation criteria, third-party validation, and outcome-linked disbursement rather than discretionary funding, thereby operationalizing sovereignty as participation in value creation rather than passive data provision (<xref ref-type="fig" rid="figure2">Figure 2</xref>).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Feedback loop of the low- and middle-income country (LMIC) data sovereignty model. A self-reinforcing cycle representing how LMICs can transition from data poverty to data sovereignty. Local institutions generate graded datasets (eg, via Medical Imaging Datasets for India), which are used by AI developers and externally validated. Models that meet performance thresholds receive outcome-linked incentives, supporting reinvestment in further data curation and strengthening the ecosystem.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e82360_fig02.png"/></fig></sec><sec id="s8"><title>Anticipated Challenges</title><sec id="s8-1"><title>Regulatory Fragmentation</title><p>Although many LMICs have enacted data-protection statutes (eg, India&#x2019;s Digital Personal Data Protection [DPDP] Act 2023), few contain explicit provisions for cross-border clinical-research data flows. Regulators should adopt a &#x201C;trust framework&#x201D; akin to the EU-US Data Privacy Framework [<xref ref-type="bibr" rid="ref19">19</xref>], enabling Commons participants to exchange deidentified data under reciprocal adequacy findings.</p></sec><sec id="s8-2"><title>Sustainability</title><p>Dataset curation is costly, requiring ongoing investment to maintain quality and support updates [<xref ref-type="bibr" rid="ref20">20</xref>]. A blended-finance approach, combining World Bank digital health loans, philanthropic grants, and small levies on AI-assisted clinical tests, has been recommended to fund the longitudinal upkeep of biomedical datasets [<xref ref-type="bibr" rid="ref21">21</xref>]. Such models are increasingly recognized as effective strategies for supporting health care data infrastructure, particularly in LMICs. Programs such as the All of Us Research Program emphasize the importance of sustained curation and highlight its role in enabling downstream research. However, specific quantitative returns on investment have not yet been established in the published literature [<xref ref-type="bibr" rid="ref22">22</xref>].</p></sec><sec id="s8-3"><title>Technical Debt</title><p>Version creep and annotation drift loom large. Federated continuous integration pipelines can enforce schema consistency only if maintainers budget for reannotation cycles and retain cloud credits for secure computing. The MIDAS dataset already schedules &#x201C;maintenance releases,&#x201D; in which new images are rescored, and governance metadata are refreshed [<xref ref-type="bibr" rid="ref12">12</xref>].</p></sec></sec><sec id="s9"><title>Global Implications</title><p>Transitioning from data scarcity to sovereignty is more than an LMIC agenda; it is a prerequisite for valid, generalizable AI everywhere. Commercial developers seek regulatory clarity, journal editors demand reproducibility, and patients deserve equity by design. A South-South Commons anchored in graded, openly documented datasets would convert what WHO terms &#x201C;an ethical imperative&#x201D; into a tangible, auditable infrastructure. HIC regulators and payers stand to benefit: models validated on Commons data will arrive already stress-tested across demographic gradients that do not exist in single-country registries, reducing the risk of catastrophic postdeployment failures.</p></sec><sec id="s10" sec-type="conclusions"><title>Conclusions</title><p>Algorithmic equity cannot be retrofitted after deployment; it must be engineered upstream through the data themselves. By institutionalizing rigorous quality grading, distributed custodianship, and value-linked incentives, LMICs can recast themselves from data supplicants to data sovereigns producing datasets that are not only locally trustworthy but also globally indispensable. The blueprint outlined here, grounded in the MIDAS experience and amplified through a South-South Data Commons, offers a practical, scalable route toward that future.</p><p>From a policy perspective, operationalizing data sovereignty does not require a wholesale system redesign. Immediate steps for LMIC governments and research agencies include formally adopting transparent dataset quality grading standards within publicly funded research programs; designating or accrediting national institutions to perform independent, privacy-preserving AI validation; and embedding gold-standard dataset performance requirements into public procurement, regulatory evaluation, and grant funding criteria. National data platforms can be made interoperable with Commons architectures by publishing versioned metadata, audit artifacts, and validation outcomes rather than raw data alone. Taken together, these measures allow sovereignty to be exercised incrementally through evaluative authority, incentive alignment, and regulatory participation while remaining compatible with cross-border collaboration and global AI development norms.</p></sec></body><back><ack><p>The authors sincerely thank the Indian Council of Medical Research (ICMR), where this work was conceptualized and completed. The authors also thank all the contributors who made this research and its publication possible. This work was enriched and made possible through the collaborative efforts of all involved. All authors declared that they had insufficient funding to support the open access publication of this manuscript, including from affiliated organizations or institutions, funding agencies, or other organizations. JMIR Publications provided article processing fee (APF) support for the publication of this article.</p></ack><notes><sec><title>Funding</title><p>The authors declare that no financial support was received for this study.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: HS</p><p>Resources: HS</p><p>Supervision: HS</p><p>Writing&#x2014;original draft: SR, HS</p><p>Writing&#x2014;review and editing: PB, SP, HS</p><p>All authors read and approved the final manuscript.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AIIMS</term><def><p>All India Institute of Medical Sciences</p></def></def-item><def-item><term id="abb2">ARTPARK</term><def><p>AI and Robotics Technology Park</p></def></def-item><def-item><term id="abb3">CC-BY-NC</term><def><p>Creative Commons Attribution-NonCommercial</p></def></def-item><def-item><term id="abb4">DPDP</term><def><p>Digital Personal Data Protection</p></def></def-item><def-item><term id="abb5">FDA</term><def><p>Food and Drug Administration</p></def></def-item><def-item><term id="abb6">HIC</term><def><p>high-income country</p></def></def-item><def-item><term id="abb7">ICMR</term><def><p>Indian Council of Medical Research</p></def></def-item><def-item><term id="abb8">IISc</term><def><p>Indian Institute of Science</p></def></def-item><def-item><term id="abb9">LMIC</term><def><p>low- and middle-income country</p></def></def-item><def-item><term id="abb10">MAIDA</term><def><p>Medical AI Data for All</p></def></def-item><def-item><term id="abb11">MIDAS</term><def><p>Medical Imaging Datasets for India</p></def></def-item><def-item><term id="abb12">MRI</term><def><p>magnetic resonance imaging</p></def></def-item><def-item><term id="abb13">NHS</term><def><p>National Health Service</p></def></def-item><def-item><term id="abb14">OECD</term><def><p>Organisation for Economic Co-operation and Development</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Muralidharan</surname><given-names>V</given-names> </name><name name-style="western"><surname>Adewale</surname><given-names>BA</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>CJ</given-names> </name><etal/></person-group><article-title>A scoping review of reporting gaps in FDA-approved AI medical devices</article-title><source>NPJ Digit Med</source><year>2024</year><month>10</month><day>3</day><volume>7</volume><issue>1</issue><fpage>273</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01270-x</pub-id><pub-id pub-id-type="medline">39362934</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Celi</surname><given-names>LA</given-names> </name><name name-style="western"><surname>Cellini</surname><given-names>J</given-names> </name><name name-style="western"><surname>Charpignon</surname><given-names>ML</given-names> </name><etal/></person-group><article-title>Sources of bias in artificial intelligence that perpetuate healthcare disparities-a global review</article-title><source>PLOS Digit Health</source><year>2022</year><month>03</month><volume>1</volume><issue>3</issue><fpage>e0000022</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000022</pub-id><pub-id pub-id-type="medline">36812532</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ciecierski-Holmes</surname><given-names>T</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>R</given-names> </name><name name-style="western"><surname>Axt</surname><given-names>M</given-names> </name><name name-style="western"><surname>Brenner</surname><given-names>S</given-names> </name><name name-style="western"><surname>Barteit</surname><given-names>S</given-names> </name></person-group><article-title>Artificial intelligence for strengthening healthcare systems in low- and middle-income countries: a systematic scoping review</article-title><source>NPJ Digit Med</source><year>2022</year><month>10</month><day>28</day><volume>5</volume><issue>1</issue><fpage>162</fpage><pub-id pub-id-type="doi">10.1038/s41746-022-00700-y</pub-id><pub-id pub-id-type="medline">36307479</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ibrahim</surname><given-names>H</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zariffa</surname><given-names>N</given-names> </name><name name-style="western"><surname>Morris</surname><given-names>AD</given-names> </name><name name-style="western"><surname>Denniston</surname><given-names>AK</given-names> </name></person-group><article-title>Health data poverty: an assailable barrier to equitable digital health care</article-title><source>Lancet Digit Health</source><year>2021</year><month>04</month><volume>3</volume><issue>4</issue><fpage>e260</fpage><lpage>e265</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(20)30317-4</pub-id><pub-id pub-id-type="medline">33678589</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Paik</surname><given-names>KE</given-names> </name><name name-style="western"><surname>Hicklen</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kaggwa</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Digital determinants of health: health data poverty amplifies existing health disparities-a scoping review</article-title><source>PLOS Digit Health</source><year>2023</year><month>10</month><volume>2</volume><issue>10</issue><fpage>e0000313</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000313</pub-id><pub-id pub-id-type="medline">37824445</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alderman</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Charalambides</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sachdeva</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Revealing transparency gaps in publicly available COVID-19 datasets used for medical artificial intelligence development-a systematic review</article-title><source>Lancet Digit Health</source><year>2024</year><month>11</month><volume>6</volume><issue>11</issue><fpage>e827</fpage><lpage>e847</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(24)00146-8</pub-id><pub-id pub-id-type="medline">39455195</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Khan</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Nath</surname><given-names>S</given-names> </name><etal/></person-group><article-title>A global review of publicly available datasets for ophthalmological imaging: barriers to access, usability, and generalisability</article-title><source>Lancet Digit Health</source><year>2021</year><month>01</month><volume>3</volume><issue>1</issue><fpage>e51</fpage><lpage>e66</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(20)30240-5</pub-id><pub-id pub-id-type="medline">33735069</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dishner</surname><given-names>KA</given-names> </name><name name-style="western"><surname>McRae-Posani</surname><given-names>B</given-names> </name><name name-style="western"><surname>Bhowmik</surname><given-names>A</given-names> </name><etal/></person-group><article-title>A survey of publicly available MRI datasets for potential use in artificial intelligence research</article-title><source>J Magn Reson Imaging</source><year>2024</year><month>02</month><volume>59</volume><issue>2</issue><fpage>450</fpage><lpage>480</lpage><pub-id pub-id-type="doi">10.1002/jmri.29101</pub-id><pub-id pub-id-type="medline">37888298</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Clifton</surname><given-names>L</given-names> </name><name name-style="western"><surname>Dung</surname><given-names>NT</given-names> </name><etal/></person-group><article-title>Mitigating machine learning bias between high income and low&#x2013;middle income countries for enhanced model fairness and generalizability</article-title><source>Sci Rep</source><year>2024</year><month>06</month><day>10</day><volume>14</volume><issue>1</issue><fpage>13318</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-64210-5</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="web"><article-title>Ethics and governance of artificial intelligence for health: guidance on large multi-modal models</article-title><source>World Health Organization</source><year>2025</year><access-date>2026-07-30</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://iris.who.int/bitstream/handle/10665/375579/9789240084759-eng.pdf?sequence=1">https://iris.who.int/bitstream/handle/10665/375579/9789240084759-eng.pdf?sequence=1</ext-link></comment></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Evertsz</surname><given-names>N</given-names> </name><name name-style="western"><surname>Bull</surname><given-names>S</given-names> </name><name name-style="western"><surname>Pratt</surname><given-names>B</given-names> </name></person-group><article-title>What constitutes equitable data sharing in global health research? A scoping review of the literature on low-income and middle-income country stakeholders&#x2019; perspectives</article-title><source>BMJ Glob Health</source><year>2023</year><month>03</month><volume>8</volume><issue>3</issue><fpage>e010157</fpage><pub-id pub-id-type="doi">10.1136/bmjgh-2022-010157</pub-id><pub-id pub-id-type="medline">36977523</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Maity</surname><given-names>D</given-names> </name><name name-style="western"><surname>Satish</surname><given-names>R</given-names> </name><name name-style="western"><surname>Jadeja</surname><given-names>DA</given-names> </name><etal/></person-group><article-title>MIDAS: a new platform for quality-graded health data for AI-enabled healthcare in India</article-title><source>Nat Med</source><year>2024</year><month>10</month><volume>30</volume><issue>10</issue><fpage>2704</fpage><lpage>2705</lpage><pub-id pub-id-type="doi">10.1038/s41591-024-03198-x</pub-id><pub-id pub-id-type="medline">39210000</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="web"><source>AIKosh</source><access-date>2026-08-04</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://aikosh.indiaai.gov.in/home">https://aikosh.indiaai.gov.in/home</ext-link></comment></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Saenz</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>E</given-names> </name><name name-style="western"><surname>Marklund</surname><given-names>H</given-names> </name><name name-style="western"><surname>Rajpurkar</surname><given-names>P</given-names> </name></person-group><article-title>The MAIDA initiative: establishing a framework for global medical-imaging data sharing</article-title><source>Lancet Digit Health</source><year>2024</year><month>01</month><volume>6</volume><issue>1</issue><fpage>e6</fpage><lpage>e8</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(23)00222-4</pub-id><pub-id pub-id-type="medline">37977999</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Lekadir</surname><given-names>K</given-names> </name><name name-style="western"><surname>Feragen</surname><given-names>A</given-names> </name><name name-style="western"><surname>Fofanah</surname><given-names>AJ</given-names> </name><etal/></person-group><article-title>FUTURE-AI: international consensus guideline for trustworthy and deployable artificial intelligence in healthcare</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 11, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2309.12325</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="web"><article-title>India Africa health sciences platform</article-title><source>Indian Council of Medical Research</source><access-date>2026-07-30</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.icmr.gov.in/icmrobject/static/icmr/dist/images/pdf/iahsp/V5_About_IAHSP.pdf">https://www.icmr.gov.in/icmrobject/static/icmr/dist/images/pdf/iahsp/V5_About_IAHSP.pdf</ext-link></comment></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fajtl</surname><given-names>J</given-names> </name><name name-style="western"><surname>Welikala</surname><given-names>RA</given-names> </name><name name-style="western"><surname>Barman</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Trustworthy evaluation of clinical AI for analysis of medical images in diverse populations</article-title><source>NEJM AI</source><year>2024</year><month>08</month><day>22</day><volume>1</volume><issue>9</issue><pub-id pub-id-type="doi">10.1056/AIoa2400353</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cushnan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Bennett</surname><given-names>O</given-names> </name><name name-style="western"><surname>Berka</surname><given-names>R</given-names> </name><etal/></person-group><article-title>An overview of the National COVID-19 Chest Imaging Database: data quality and cohort analysis</article-title><source>Gigascience</source><year>2021</year><month>11</month><day>25</day><volume>10</volume><issue>11</issue><fpage>giab076</fpage><pub-id pub-id-type="doi">10.1093/gigascience/giab076</pub-id><pub-id pub-id-type="medline">34849869</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tschider</surname><given-names>C</given-names> </name><name name-style="western"><surname>Compagnucci</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Minssen</surname><given-names>T</given-names> </name></person-group><article-title>The new EU-US data protection framework&#x2019;s implications for healthcare</article-title><source>J Law Biosci</source><year>2024</year><volume>11</volume><issue>2</issue><fpage>lsae022</fpage><pub-id pub-id-type="doi">10.1093/jlb/lsae022</pub-id><pub-id pub-id-type="medline">39346780</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alberto</surname><given-names>IR</given-names> </name><name name-style="western"><surname>Alberto</surname><given-names>NR</given-names> </name><name name-style="western"><surname>Ghosh</surname><given-names>AK</given-names> </name><etal/></person-group><article-title>The impact of commercial health datasets on medical research and health-care algorithms</article-title><source>Lancet Digit Health</source><year>2023</year><month>05</month><volume>5</volume><issue>5</issue><fpage>e288</fpage><lpage>e294</lpage><pub-id pub-id-type="doi">10.1016/S2589-7500(23)00025-0</pub-id><pub-id pub-id-type="medline">37100543</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="web"><article-title>The global plan to end TB 2023-2030</article-title><source>Stop TB Partnership</source><year>2022</year><access-date>2026-07-30</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.stoptb.org/sites/default/files/documents/global_plan_to_end_tb_2023-2030%20%283%29.pdf">https://www.stoptb.org/sites/default/files/documents/global_plan_to_end_tb_2023-2030%20%283%29.pdf</ext-link></comment></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>K</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>E</given-names> </name><name name-style="western"><surname>Theodorou</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Characterizing the clinical adoption of medical AI devices through U.S. insurance claims</article-title><source>NEJM AI</source><year>2024</year><volume>1</volume><issue>1</issue><pub-id pub-id-type="doi">10.1056/AIoa2300030</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Four-domain Dataset Quality Matrix.</p><media xlink:href="medinform_v14i1e82360_app1.docx" xlink:title="DOCX File, 17 KB"/></supplementary-material></app-group></back></article>