<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Data</journal-id><journal-id journal-id-type="publisher-id">data</journal-id><journal-id journal-id-type="index">25</journal-id><journal-title>JMIR Data</journal-title><abbrev-journal-title>JMIR Data</abbrev-journal-title><issn pub-type="epub">2819-4497</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v7i1e85688</article-id><article-id pub-id-type="doi">10.2196/85688</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Data Availability for Critical Appraisal Tools for Evaluating Artificial Intelligence in Clinical Studies: Dataset for a Scoping Review</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Cabello L&#x00F3;pez</surname><given-names>Juan Bautista</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ruiz Garc&#x00ED;a</surname><given-names>Vicente</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Torralba</surname><given-names>Miguel</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Maldonado Fernandez</surname><given-names>Miguel</given-names></name><degrees>MSc, MPH, MD, PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>&#x00DA;beda-Carrillo</surname><given-names>Marimar</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ansuategi</surname><given-names>Eukene</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ramos-Ruperto</surname><given-names>Luis</given-names></name><degrees>MSc, MD</degrees><xref ref-type="aff" rid="aff7">7</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Emparanza</surname><given-names>Jos&#x00E9; Ignacio</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff8">8</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Urreta-Barallobre</surname><given-names>Iratxe</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff8">8</xref><xref ref-type="aff" rid="aff9">9</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Iglesias Gaspar</surname><given-names>Mar&#x00ED;a-Teresa</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff8">8</xref><xref ref-type="aff" rid="aff10">10</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Pijoan Zubizarreta</surname><given-names>Jos&#x00E9; Ignacio</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff11">11</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Burls</surname><given-names>Amanda J</given-names></name><degrees>MBBS, MSc</degrees><xref ref-type="aff" rid="aff12">12</xref></contrib></contrib-group><aff id="aff1"><institution>CASP Espa&#x00F1;a &#x0026; Grupo Medicine AI</institution><addr-line>Alicante</addr-line><country>Spain</country></aff><aff id="aff2"><institution>Servicio de Hospitalizaci&#x00F3;n a Domicilio, Hospital Universitari i Polit&#x00E8;cnic La Fe</institution><addr-line>Valencia</addr-line><country>Spain</country></aff><aff id="aff3"><institution>Servicio de Medicina Interna (Infecciosas), Hospital Universitario de Guadalajara</institution><addr-line>Guadalajara</addr-line><country>Spain</country></aff><aff id="aff4"><institution>Hospital Vital &#x00C1;lvarez-Buylla</institution><addr-line>Calle Vistalegre, 2</addr-line><addr-line>Mieres</addr-line><country>Spain</country></aff><aff id="aff5"><institution>Library Service, Donostia University Hospital, Osakidetza Basque Health Service</institution><addr-line>San Sebasti&#x00E1;n</addr-line><country>Spain</country></aff><aff id="aff6"><institution>Biblioteca virtual de salud de Euskadi</institution><addr-line>Vitoria</addr-line><country>Spain</country></aff><aff id="aff7"><institution>Hospital Universitario La Paz</institution><addr-line>Madrid</addr-line><country>Spain</country></aff><aff id="aff8"><institution>Clinical Epidemiology Unit, Donostia University Hospital, Osakidetza Basque Health Service</institution><addr-line>San Sebasti&#x00E1;n</addr-line><country>Spain</country></aff><aff id="aff9"><institution>CIBER de Epidemiolog&#x00ED;a y Salud P&#x00FA;blica (CIBERESP), Instituto de Salud Carlos III</institution><addr-line>Madrid</addr-line><country>Spain</country></aff><aff id="aff10"><institution>Faculty of Health Sciences, University of Deusto</institution><addr-line>San Sebasti&#x00E1;n</addr-line><country>Spain</country></aff><aff id="aff11"><institution>Hospital de Cruces</institution><addr-line>Bilbao</addr-line><country>Spain</country></aff><aff id="aff12"><institution>City St George's, University of London</institution><addr-line>London</addr-line><country>United Kingdom</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Mavragani</surname><given-names>Amaryllis</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Conti</surname><given-names>Andrea</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Kim</surname><given-names>Seongsoon</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Miguel Maldonado Fernandez, MSc, MPH, MD, PhD, Hospital Vital &#x00C1;lvarez-Buylla, Calle Vistalegre, 2, Mieres, 33611, Spain, 34 652835133; <email>maldonado2000@gmail.com</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>28</day><month>8</month><year>2026</year></pub-date><volume>7</volume><elocation-id>e85688</elocation-id><history><date date-type="received"><day>13</day><month>10</month><year>2025</year></date><date date-type="rev-recd"><day>27</day><month>04</month><year>2026</year></date><date date-type="accepted"><day>03</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Juan Bautista Cabello L&#x00F3;pez, Vicente Ruiz Garc&#x00ED;a, Miguel Torralba, Miguel Maldonado Fernandez, Mar&#x00ED;a del Mar &#x00DA;beda-Carrillo, Eukene Ansuategi, Luis Ramos-Ruperto, Jos&#x00E9; Ignacio Emparanza, Iratxe Urreta-Barallobre, Mar&#x00ED;a-Teresa Iglesias Gaspar, Jos&#x00E9; Ignacio Pijoan Zubizarreta, Amanda J Burls. Originally published in JMIR Data (<ext-link ext-link-type="uri" xlink:href="https://data.jmir.org">https://data.jmir.org</ext-link>), 28.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">http://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Data, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://data.jmir.org/">https://data.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://data.jmir.org/2026/1/e85688"/><abstract><sec><title>Background</title><p>AI is increasingly used in clinical research, generating a growing need for robust critical appraisal tools to evaluate methodological quality, reporting standards, and potential biases. While traditional instruments exist for conventional clinical studies, specific tools designed for AI-based research are still emerging.</p></sec><sec><title>Objective</title><p>This dataset accompanies a scoping review that aimed to identify and describe existing critical appraisal tools, reporting frameworks, and bias classification systems applicable to clinical studies using AI, including chatbot-based interventions.</p></sec><sec sec-type="methods"><title>Methods</title><p>We systematically searched MEDLINE, Embase, CINAHL, PsycINFO, and IEEE Xplore from inception to April 2024. Eligible studies included those proposing or using tools for critical appraisal, reporting, quality assessment, or risk of bias in AI-related clinical research. Screening and extraction followed JBI and PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Scoping Reviews) recommendations. Data extraction combined human review with a supervised GPT-4o&#x2013;based retrieval-augmented generation (RAG) process to enhance transparency and reproducibility. All AI-assisted outputs were verified independently by two reviewers.</p></sec><sec sec-type="results"><title>Results</title><p>Seventy records were included: 46 reporting guidelines (comprising 26 guides for reporting AI studies, 16 critical appraisal tools, 2 quality assessment instruments, and 2 risk-of-bias tools), 9 bias classification or bias mitigation studies, and 15 chatbot evaluation studies. All datasets, extraction templates, and RAG prompts are publicly available to facilitate validation and reuse.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This dataset provides a comprehensive overview of critical appraisal and reporting tools for AI-based clinical research. It may support the development of standardized evaluation frameworks and promote transparency in future AI-assisted health studies.</p></sec><sec><title>Trial Registration</title><p>OSF Registries 10.17605/OSF.IO/ETYDS; https://osf.io/etyds/overview</p></sec></abstract><kwd-group><kwd>artificial intelligence</kwd><kwd>critical appraisal</kwd><kwd>bias</kwd><kwd>data sharing</kwd><kwd>scoping review</kwd><kwd>reporting guidelines</kwd><kwd>chatbot</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>The use of predictive or generative AI in health research is rapidly growing [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref6">6</xref>]. As in traditional clinical studies, the methods used in AI-assisted studies can introduce systematic errors. The translation of AI-assisted evidence into clinical practice and research requires critical appraisal tools for clinical decision-makers and researchers [<xref ref-type="bibr" rid="ref7">7</xref>-<xref ref-type="bibr" rid="ref10">10</xref>]. We carried out a scoping review [<xref ref-type="bibr" rid="ref11">11</xref>] to identify existing tools for the critical appraisal of clinical studies that use AI and to examine the concepts and domains these tools explore. Our research question is framed using the PCC (population, concept, and context) framework [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]: the population includes clinical studies using AI; the concept refers to tools for critical appraisal and associated constructs such as quality, reporting, validity, risk of bias, and applicability; and the context is clinical practice. In addition, bias classification and chatbot assessment studies were included.</p><p>A total of 70 records were included in the review, comprising a heterogeneous group. Of these, 46 were tools (26 guides for reporting AI studies, 16 tools for critical appraisal, 2 tools for study quality, and 2 tools for risk of bias), 9 were papers focused on bias classification or mitigation, and 15 were chatbot studies (6 chatbot assessment studies and 9 systematic reviews of chatbot studies).</p><p>Just as traditional health research may include systematic errors that lead to biased or nongeneralizable results, so AI methods can introduce their own systematic errors at the design, data collection, training, or evaluation stages, which threaten the validity and reliability of AI models&#x2019; data analyses, findings, and conclusions. Such errors can arise from a number of different sources, including, but not limited to, flawed data, biased algorithms, and incorrect training. Health care decision-makers therefore need to be able to critically appraise AI studies to detect these problems so they can assess the certainty and relevance of the presented evidence.</p><p>Access to this dataset will help systematize the available tools and allow other researchers to compare and select appropriate ones. The dataset will also be of use for teaching purposes and risk-of-bias evaluation. In addition, we believe that it is good scientific practice to share data.</p><p>Our objective is to make the primary data used in our scoping review publicly available.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>We searched medical and engineering databases (MEDLINE, Embase, CINAHL, PsycINFO, and IEEE) from inception to April 2024. We included primary clinical research that used tools for critical appraisal. Classic reviews and systematic reviews were included in the first phase of screening and used to identify new tools by forward snowballing. They were excluded in the second phase. We excluded nonhuman, computer, and mathematical research and letters, opinion papers, and editorials. We used Rayyan for screening [<xref ref-type="bibr" rid="ref14">14</xref>].</p><p>The protocol was previously registered with OSF [<xref ref-type="bibr" rid="ref15">15</xref>]. We adhered to the PRISMA-ScR (Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Scoping Reviews; <xref ref-type="supplementary-material" rid="app8">Checklist 1</xref>) [<xref ref-type="bibr" rid="ref16">16</xref>] and the PRISMA-S (Preferred Reporting Items for Systematic Reviews Literature Search Extension) for reporting literature in systematic reviews [<xref ref-type="bibr" rid="ref17">17</xref>].</p><p>Data extraction for tools and bias was performed according to JBI recommendations [<xref ref-type="bibr" rid="ref18">18</xref>], and each study was rated by two observers. Discrepancies were resolved by discussion and consensus in an iterative process. (Data extraction tables and guidance are shown in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.)</p><p>Data extraction for chatbot studies used a hybrid approach, combining the active involvement of a researcher with a retrieval-augmented generation (RAG) approach using custom ChatGPT instances based on the GPT-4o model, accessed directly through the ChatGPT web interface, with default model parameters (the RAG prompt template in Spanish is available in Spanish in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> and in English in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>). Each PDF was uploaded and analyzed as an independent prompt, and all extracted information was subsequently reviewed by the primary investigator and independently validated by a second investigator to minimize hallucinations and ensure that the large language model was used solely as a facilitative extraction tool. No sensitive data were exposed. To promote transparency and reproducibility, the exact prompts used in the RAG process are incorporated into this dataset.</p><p>As an example to explain the process, the systematic review approach with RAG (in Spanish) was used to extract data from Oh et al [<xref ref-type="bibr" rid="ref19">19</xref>]. A PDF was uploaded to ChatGPT (GPT-4o), and the output was provided in a PDF file (<xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>). The result was incorporated as a column in the rev_sis extraction XLS file, which can be found in column H in the dataset [<xref ref-type="bibr" rid="ref11">11</xref>]. This procedure was repeated for all systematic review articles and for all primary research articles (rev_sis extraction and primary studies extraction, respectively). Both XLS files were transposed from columns to rows (matrix transposition) to build the final XLS files (S reviews table draft and primary studies table draft, respectively). The results were translated from Spanish to English by the reviewers. Both table drafts were reviewed and their results compared with the actual papers by LR-R and afterward verified by JBCL. The final tables were synthesized as shown in the scoping review, where they appear as Tables 4 and 5 [<xref ref-type="bibr" rid="ref11">11</xref>].</p><p>We identified 4392 records in the selected databases and registries. After eliminating 470 duplicates, 3922 records were screened by title and abstract, and 3803 were excluded. The remaining 119 underwent full-text screening, and 59 were excluded. The reasons for exclusion were as follows: 50 studies were systematic reviews, 7 met the exclusion criteria, and 2 did not meet the inclusion criteria. Full details are available in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>. Of the 50 systematic reviews, 42 used specific AI tools to assess the quality of the reviewed studies, and the tools retrieved were incorporated into the &#x201C;records identified via other methods&#x201D; category.</p><p>Twelve studies were identified in the EQUATOR Network library, and 4 additional studies were obtained from experts and organizations; therefore, 58 records were identified by other methods. Forty-eight of these were already captured in the 60 included studies from the search of electronic databases, leaving 10 additional studies to be included.</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>Institutional review board approval was not applicable to this study and dataset. The scoping review built on this dataset has been accepted for publication in the <italic>Journal of Medical Internet Research</italic>.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><p>We identified 4392 records in databases and registries. After the selection process and inclusion of additional studies described in the Methods section, a total of 70 studies were included in the review (see scoping review [<xref ref-type="bibr" rid="ref11">11</xref>]; <xref ref-type="fig" rid="figure1">Figure 1</xref>). Forty-six records reported tools (26 guidelines, 16 tools for critical appraisal, 2 tools for study quality, and 2 tools for risk of bias), and 9 were focused on bias classification or mitigation (<xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>). In addition, 15 records were chatbot studies, comprising 6 chatbot assessment studies and 9 systematic reviews of chatbot studies (<xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>PRISMA (Preferred Reporting Items for Systematic Reviews and Meta-Analyses) flow diagram. DNMIC: does not meet inclusion criteria; MEC: methodological/editorial/commentary; SR-NSAIT: systematic review&#x2014;nonstudies assessing AI tools; SR-SAIT: systematic review&#x2014;studies assessing AI tools.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="data_v7i1e85688_fig01.png"/></fig></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Overview</title><p>As explained in our main paper [<xref ref-type="bibr" rid="ref11">11</xref>], we conducted a comprehensive scoping review and identified 70 papers corresponding to the 3 proposed areas of research: tools for critical appraisal, bias and bias mitigation, and chatbot assessment studies. Although critical appraisal tools are the main focus of the review, types of AI bias were also included in the review because the validity (or absence of bias) is an important component of critical appraisal. Chatbot studies were included in the review because they represent an important recent, disruptive technology. The three areas together map the current landscape of evidence in the critical appraisal of clinical AI studies.</p><p>We selected critical appraisal as the main domain for the review because it is a wider and more inclusive concept than risk of bias, quality, or reporting, and it is more related to clinical practice. This decision required a change to the published protocol and was made after discussion.</p><p>Reporting guidelines are essential for authors in writing studies and for editors in maintaining consistency across publications. Critical appraisal tools are more focused on making judgments about the validity and applicability of evidence, and they are mainly intended for dissemination and teaching purposes. A paper may be of no use in a clinical setting, even if it is perfectly reported and its data are valid. Finally, both quality and risk of bias are precise concepts, and their tools are complex and designed as far as possible to avoid inconsistencies. Such tools are more suitable for use in research syntheses. Nevertheless, reporting, critical appraisal, risk of bias, and quality form a cluster of closely related constructs with overlapping areas.</p><p>Adequate reporting varies by the structure and type of study and is not only an editorial requirement but a part of study quality. Obviously, good reporting is a precondition to assess study quality, but there is also empirical evidence that some reporting flaws (and some nonreporting flaws) are associated with bias in the effect estimation [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. Therefore, exploring reporting is essential to judge the validity of any study, as it facilitates study replication, risk-of-bias or quality assessments, interpretation of the results, and judgment of the value and applicability of the results in real clinical settings for individualized or collective decisions. It is also needed for the inclusion and assessment of studies in systematic reviews and for the evaluation of systematic reviews. Therefore, it is part of the critical appraisal process [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref22">22</xref>].</p><p>The overlap of reporting and critical appraisal was a source of inconsistency between raters when classifying papers in this scoping review. Iterative discussions were necessary to reach consensus. The most important criteria we used to classify papers within the critical appraisal category were the relevance of the question in the clinical context and a clear intent to help with applicability.</p><p>On the other hand, chatbot assessment studies are heterogeneous and inconsistent in their design, analysis, and reporting, so we used ChatGPT (GPT-4o) for data extraction; however, all outputs were independently reviewed by two authors against the original articles, and no major corrections to the extracted information were required. Therefore, we believe this was a valid and consistent procedure for data extraction. Nevertheless, we include the RAG and instructions for matrix transposition to enable other researchers to replicate the process.</p><p>A recent systematic review [<xref ref-type="bibr" rid="ref23">23</xref>] synthesized reporting guidelines as well as tools for basic and laboratory research. However, the search was conducted only through 2022. The available reporting guidelines should be harmonized, and the review would benefit from being updated or reformulated from a clinical standpoint.</p><p>There are some limitations of this scoping review. First, we used a general search that included all of our study&#x2019;s questions; it was not specifically designed to search for bias and bias mitigation or for chatbot assessment. However, the absence of MeSH for chatbot studies and the heterogeneity of objectives, research questions, study design, devices, and analyses make searches for this type of study difficult. In addition, a potential limitation lies in the methods used to organize data extraction, as the application of large language models in evidence synthesis is novel, and formal standards for their integration are still under development. Finally, this field is evolving very quickly, so many conclusions drawn from existing evidence have a limited period of validity.</p><p>Critical appraisal tools are enormously varied, with different nuances and approaches, so selecting one can be very challenging. We believe that this topic deserves a qualitative synthesis to clarify the key elements for choosing the appropriate tool.</p><p>New risk-of-bias tools for AI in prognosis and diagnosis (such as QUADAS-AI [Quality Assessment of Diagnostic Accuracy Studies Using AI] and PROBAST+AI [Prediction Model Risk of Bias Assessment Tool]) and the PRISMA-AI (Preferred Reporting Items for Systematic Reviews and Meta-Analyses&#x2014;Artificial Intelligence Extension) for systematic reviews are expected to be published, as is CHART (Chatbot Assessment Reporting Tool), a tool for reporting chatbot assessment studies. The AI extensions of other classic tools, such as the Cochrane risk-of-bias tool and ROBINS-I (Risk of Bias in Non-Randomized Studies of Interventions), among others, should be considered. On the other hand, the development of standards for the design, reporting, and assessment of chatbot assessment studies and chatbot health-advising studies is a clear gap in our toolbox and needs to be addressed.</p><p>In clinical practice, it is important to clarify the appropriate selection of tools for critical appraisal; furthermore, it is essential to develop teaching strategies to promote skills for the critical appraisal of AI-assisted studies, including understanding the types of bias to be tackled.</p></sec><sec id="s4-2"><title>Conclusions</title><p>This dataset may be useful for other researchers who want to corroborate or further develop our results about critical appraisal tools for clinical studies that use AI.</p></sec></sec></body><back><ack><p>Although we used ChatGPT (GPT-4o) for data extraction, we did not use generative AI for manuscript production.</p></ack><notes><sec><title>Funding</title><p>No external financial support or grants were received from any public, commercial, or not-for-profit entities for the research, authorship, or publication of this article.</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">CHART</term><def><p>Chatbot Assessment Reporting Tool</p></def></def-item><def-item><term id="abb2">PCC</term><def><p>population, concept, and context</p></def></def-item><def-item><term id="abb3">PRISMA-AI</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses&#x2014;Artificial Intelligence Extension</p></def></def-item><def-item><term id="abb4">PRISMA-S</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses Literature Search Extension</p></def></def-item><def-item><term id="abb5">PRISMA-ScR</term><def><p>Preferred Reporting Items for Systematic Reviews and Meta-Analyses Extension for Scoping Reviews</p></def></def-item><def-item><term id="abb6">PROBAST+AI</term><def><p>Prediction Model Risk of Bias Assessment Tool</p></def></def-item><def-item><term id="abb7">QUADAS-AI</term><def><p>Quality Assessment of Diagnostic Accuracy Studies Using AI</p></def></def-item><def-item><term id="abb8">RAG</term><def><p>retrieval-augmented generation</p></def></def-item><def-item><term id="abb9">ROBINS-I</term><def><p>Risk of Bias in Non-Randomized Studies of Interventions</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kaul</surname><given-names>V</given-names> </name><name name-style="western"><surname>Enslin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gross</surname><given-names>SA</given-names> </name></person-group><article-title>History of artificial intelligence in medicine</article-title><source>Gastrointest Endosc</source><year>2020</year><month>10</month><volume>92</volume><issue>4</issue><fpage>807</fpage><lpage>812</lpage><pub-id pub-id-type="doi">10.1016/j.gie.2020.06.040</pub-id><pub-id pub-id-type="medline">32565184</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kohane</surname><given-names>IS</given-names> </name></person-group><article-title>Injecting artificial intelligence into medicine</article-title><source>NEJM AI</source><year>2024</year><month>01</month><volume>1</volume><issue>1</issue><pub-id pub-id-type="doi">10.1056/AIe2300197</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jayakumar</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sounderajah</surname><given-names>V</given-names> </name><name name-style="western"><surname>Normahani</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Quality assessment standards in artificial intelligence diagnostic accuracy systematic reviews: a meta-research study</article-title><source>NPJ Digit Med</source><year>2022</year><month>01</month><day>27</day><volume>5</volume><issue>1</issue><fpage>11</fpage><pub-id pub-id-type="doi">10.1038/s41746-021-00544-y</pub-id><pub-id pub-id-type="medline">35087178</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Quirk</surname><given-names>J</given-names> </name><name name-style="western"><surname>Mac Donnchadha</surname><given-names>C</given-names> </name><name name-style="western"><surname>Vaantaja</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Future implications of artificial intelligence in lung cancer screening: a systematic review</article-title><source>BJR Open</source><year>2024</year><month>10</month><day>15</day><volume>6</volume><issue>1</issue><fpage>tzae035</fpage><pub-id pub-id-type="doi">10.1093/bjro/tzae035</pub-id><pub-id pub-id-type="medline">39444460</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fleuren</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Klausch</surname><given-names>TLT</given-names> </name><name name-style="western"><surname>Zwager</surname><given-names>CL</given-names> </name><etal/></person-group><article-title>Machine learning for the prediction of sepsis: a systematic review and meta-analysis of diagnostic test accuracy</article-title><source>Intensive Care Med</source><year>2020</year><month>03</month><volume>46</volume><issue>3</issue><fpage>383</fpage><lpage>400</lpage><pub-id pub-id-type="doi">10.1007/s00134-019-05872-y</pub-id><pub-id pub-id-type="medline">31965266</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chakraborty</surname><given-names>C</given-names> </name><name name-style="western"><surname>Pal</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bhattacharya</surname><given-names>M</given-names> </name><name name-style="western"><surname>Dash</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>SS</given-names> </name></person-group><article-title>Overview of chatbots with special emphasis on artificial intelligence-enabled ChatGPT in medical science</article-title><source>Front Artif Intell</source><year>2023</year><volume>6</volume><fpage>1237704</fpage><pub-id pub-id-type="doi">10.3389/frai.2023.1237704</pub-id><pub-id pub-id-type="medline">38028668</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barker</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Stone</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Sears</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Revising the JBI quantitative critical appraisal tools to improve their applicability: an overview of methods and the development process</article-title><source>JBI Evid Synth</source><year>2023</year><month>03</month><day>1</day><volume>21</volume><issue>3</issue><fpage>478</fpage><lpage>493</lpage><pub-id pub-id-type="doi">10.11124/JBIES-22-00125</pub-id><pub-id pub-id-type="medline">36121230</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Moher</surname><given-names>D</given-names> </name></person-group><article-title>Reporting guidelines: doing better for readers</article-title><source>BMC Med</source><year>2018</year><month>12</month><day>14</day><volume>16</volume><issue>1</issue><fpage>233</fpage><pub-id pub-id-type="doi">10.1186/s12916-018-1226-0</pub-id><pub-id pub-id-type="medline">30545364</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ibrahim</surname><given-names>H</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Rivera</surname><given-names>SC</given-names> </name><etal/></person-group><article-title>Reporting guidelines for clinical trials of artificial intelligence interventions: the SPIRIT-AI and CONSORT-AI guidelines</article-title><source>Trials</source><year>2021</year><month>01</month><day>6</day><volume>22</volume><issue>1</issue><fpage>11</fpage><pub-id pub-id-type="doi">10.1186/s13063-020-04951-6</pub-id><pub-id pub-id-type="medline">33407780</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Crossnohere</surname><given-names>NL</given-names> </name><name name-style="western"><surname>Elsaid</surname><given-names>M</given-names> </name><name name-style="western"><surname>Paskett</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bose-Brill</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bridges</surname><given-names>JFP</given-names> </name></person-group><article-title>Guidelines for artificial intelligence in medicine: literature review and content analysis of frameworks</article-title><source>J Med Internet Res</source><year>2022</year><month>08</month><day>25</day><volume>24</volume><issue>8</issue><fpage>e36823</fpage><pub-id pub-id-type="doi">10.2196/36823</pub-id><pub-id pub-id-type="medline">36006692</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cabello</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Ruiz Garcia</surname><given-names>V</given-names> </name><name name-style="western"><surname>Torralba</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Critical appraisal tools for evaluating artificial intelligence in clinical studies: scoping review</article-title><source>J Med Internet Res</source><year>2025</year><month>12</month><day>8</day><volume>27</volume><fpage>e77110</fpage><pub-id pub-id-type="doi">10.2196/77110</pub-id><pub-id pub-id-type="medline">41359958</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Levac</surname><given-names>D</given-names> </name><name name-style="western"><surname>Colquhoun</surname><given-names>H</given-names> </name><name name-style="western"><surname>O&#x2019;Brien</surname><given-names>KK</given-names> </name></person-group><article-title>Scoping studies: advancing the methodology</article-title><source>Implement Sci</source><year>2010</year><month>09</month><day>20</day><volume>5</volume><issue>1</issue><fpage>69</fpage><pub-id pub-id-type="doi">10.1186/1748-5908-5-69</pub-id><pub-id pub-id-type="medline">20854677</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="book"><person-group person-group-type="editor"><name name-style="western"><surname>Aromataris</surname><given-names>E</given-names> </name><name name-style="western"><surname>Lockwood</surname><given-names>C</given-names> </name><name name-style="western"><surname>Porritt</surname><given-names>K</given-names> </name><name name-style="western"><surname>Pilla</surname><given-names>B</given-names> </name><name name-style="western"><surname>Jordan</surname><given-names>Z</given-names> </name></person-group><source>JBI Manual for Evidence Synthesis</source><year>2024</year><publisher-name>JBI</publisher-name><pub-id pub-id-type="doi">10.46658/JBIMES-24-01</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ouzzani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hammady</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fedorowicz</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Elmagarmid</surname><given-names>A</given-names> </name></person-group><article-title>Rayyan&#x2014;a web and mobile app for systematic reviews</article-title><source>Syst Rev</source><year>2016</year><month>12</month><day>5</day><volume>5</volume><issue>1</issue><fpage>210</fpage><pub-id pub-id-type="doi">10.1186/s13643-016-0384-4</pub-id><pub-id pub-id-type="medline">27919275</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="web"><article-title>Critical appraisal tool for artificial intelligence clinical studies. a scoping review</article-title><source>Open Science Framework</source><year>2024</year><month>04</month><day>18</day><access-date>2026-09-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://osf.io/etyds/overview">https://osf.io/etyds/overview</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tricco</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Lillie</surname><given-names>E</given-names> </name><name name-style="western"><surname>Zarin</surname><given-names>W</given-names> </name><etal/></person-group><article-title>PRISMA extension for scoping reviews (PRISMA-ScR): checklist and explanation</article-title><source>Ann Intern Med</source><year>2018</year><month>10</month><day>2</day><volume>169</volume><issue>7</issue><fpage>467</fpage><lpage>473</lpage><pub-id pub-id-type="doi">10.7326/M18-0850</pub-id><pub-id pub-id-type="medline">30178033</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rethlefsen</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Kirtley</surname><given-names>S</given-names> </name><name name-style="western"><surname>Waffenschmidt</surname><given-names>S</given-names> </name><etal/></person-group><article-title>PRISMA-S: an extension to the PRISMA Statement for Reporting Literature Searches in Systematic Reviews</article-title><source>Syst Rev</source><year>2021</year><month>01</month><day>26</day><volume>10</volume><issue>1</issue><fpage>39</fpage><pub-id pub-id-type="doi">10.1186/s13643-020-01542-z</pub-id><pub-id pub-id-type="medline">33499930</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pollock</surname><given-names>D</given-names> </name><name name-style="western"><surname>Peters</surname><given-names>MDJ</given-names> </name><name name-style="western"><surname>Khalil</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Recommendations for the extraction, analysis, and presentation of results in scoping reviews</article-title><source>JBI Evid Synth</source><year>2023</year><month>03</month><day>1</day><volume>21</volume><issue>3</issue><fpage>520</fpage><lpage>532</lpage><pub-id pub-id-type="doi">10.11124/JBIES-22-00123</pub-id><pub-id pub-id-type="medline">36081365</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Oh</surname><given-names>YJ</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Fukuoka</surname><given-names>Y</given-names> </name></person-group><article-title>A systematic review of artificial intelligence chatbots for promoting physical activity, healthy diet, and weight loss</article-title><source>Int J Behav Nutr Phys Act</source><year>2021</year><month>12</month><day>11</day><volume>18</volume><issue>1</issue><fpage>160</fpage><pub-id pub-id-type="doi">10.1186/s12966-021-01224-6</pub-id><pub-id pub-id-type="medline">34895247</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dechartres</surname><given-names>A</given-names> </name><name name-style="western"><surname>Trinquart</surname><given-names>L</given-names> </name><name name-style="western"><surname>Faber</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ravaud</surname><given-names>P</given-names> </name></person-group><article-title>Empirical evaluation of which trial characteristics are associated with treatment effect estimates</article-title><source>J Clin Epidemiol</source><year>2016</year><month>09</month><volume>77</volume><fpage>24</fpage><lpage>37</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2016.04.005</pub-id><pub-id pub-id-type="medline">27140444</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dwan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Altman</surname><given-names>DG</given-names> </name><name name-style="western"><surname>Clarke</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Evidence for the selective reporting of analyses and discrepancies in clinical trials: a systematic review of cohort studies of clinical trials</article-title><source>PLoS Med</source><year>2014</year><month>06</month><volume>11</volume><issue>6</issue><fpage>e1001666</fpage><pub-id pub-id-type="doi">10.1371/journal.pmed.1001666</pub-id><pub-id pub-id-type="medline">24959719</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Simera</surname><given-names>I</given-names> </name><name name-style="western"><surname>Moher</surname><given-names>D</given-names> </name><name name-style="western"><surname>Hirst</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hoey</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schulz</surname><given-names>KF</given-names> </name><name name-style="western"><surname>Altman</surname><given-names>DG</given-names> </name></person-group><article-title>Transparent and accurate reporting increases reliability, utility, and impact of your research: reporting guidelines and the EQUATOR Network</article-title><source>BMC Med</source><year>2010</year><month>04</month><day>26</day><volume>8</volume><issue>1</issue><fpage>24</fpage><pub-id pub-id-type="doi">10.1186/1741-7015-8-24</pub-id><pub-id pub-id-type="medline">20420659</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kolbinger</surname><given-names>FR</given-names> </name><name name-style="western"><surname>Veldhuizen</surname><given-names>GP</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Truhn</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kather</surname><given-names>JN</given-names> </name></person-group><article-title>Reporting guidelines in medical artificial intelligence: a systematic review and meta-analysis</article-title><source>Commun Med (Lond)</source><year>2024</year><month>04</month><day>11</day><volume>4</volume><issue>1</issue><fpage>71</fpage><pub-id pub-id-type="doi">10.1038/s43856-024-00492-0</pub-id><pub-id pub-id-type="medline">38605106</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Data extraction template.</p><media xlink:href="data_v7i1e85688_app1.docx" xlink:title="DOCX File, 22 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>ChatGPT retrieval-augmented generation prompt template (Spanish).</p><media xlink:href="data_v7i1e85688_app2.docx" xlink:title="DOCX File, 17 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>ChatGPT retrieval-augmented generation prompt template (English).</p><media xlink:href="data_v7i1e85688_app3.docx" xlink:title="DOCX File, 19 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Example of ChatGPT output.</p><media xlink:href="data_v7i1e85688_app4.pdf" xlink:title="PDF File, 278 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Exclusions after screening.</p><media xlink:href="data_v7i1e85688_app5.docx" xlink:title="DOCX File, 47 KB"/></supplementary-material><supplementary-material id="app6"><label>Multimedia Appendix 6</label><p>Included studies reporting critical appraisal tools and bias mitigation.</p><media xlink:href="data_v7i1e85688_app6.xlsx" xlink:title="XLSX File, 159 KB"/></supplementary-material><supplementary-material id="app7"><label>Multimedia Appendix 7</label><p>Included chatbot studies.</p><media xlink:href="data_v7i1e85688_app7.xlsx" xlink:title="XLSX File, 31 KB"/></supplementary-material><supplementary-material id="app8"><label>Checklist 1 PRISMA-ScR checklist</label><media xlink:href="data_v7i1e85688_app8.docx" xlink:title="DOCX File, 111 KB"/></supplementary-material></app-group></back></article>