<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JC</journal-id>
      <journal-id journal-id-type="nlm-ta">JMIR Cancer</journal-id>
      <journal-title>JMIR Cancer</journal-title>
      <issn pub-type="epub">2369-1999</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v12i1e76471</article-id>
      <article-id pub-id-type="pmid">42809848</article-id>
      <article-id pub-id-type="doi">10.2196/76471</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Assessing the Accuracy, Completeness, and Reference Quality of GPT-4 and Google for Gynecologic Cancer Information: Comparative Quantitative Study</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Bender</surname>
            <given-names>Jackie</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Bramblet</surname>
            <given-names>Rachel</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Matsuda</surname>
            <given-names>Shinichi</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>García-Barragán</surname>
            <given-names>Álvaro</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Smick</surname>
            <given-names>Alexandra H</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution>Gynecologic Oncology</institution>
            <institution>University of California, Los Angeles</institution>
            <addr-line>100 Medical Plaza Suite 383</addr-line>
            <addr-line>Los Angeles, CA, 90095</addr-line>
            <country>United States</country>
            <phone>1 310 794 7274</phone>
            <email>asmick14@gmail.com</email>
          </address>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-6396-506X</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>Ayoola</surname>
            <given-names>Martins</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-2323-1923</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Rajpara</surname>
            <given-names>Rajeshree</given-names>
          </name>
          <degrees>BS</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0003-9398-5522</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Lazo</surname>
            <given-names>Isabel</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-2831-2631</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Chase</surname>
            <given-names>Dana M</given-names>
          </name>
          <degrees>MD</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-1073-8688</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Gynecologic Oncology</institution>
        <institution>University of California, Los Angeles</institution>
        <addr-line>Los Angeles, CA</addr-line>
        <country>United States</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>Undergraduate College</institution>
        <institution>University of California, Los Angeles</institution>
        <addr-line>Los Angeles, CA</addr-line>
        <country>United States</country>
      </aff>
      <aff id="aff3">
        <label>3</label>
        <institution>Gynecologic Oncology</institution>
        <institution>Olive View-UCLA Medical Center</institution>
        <addr-line>Sylmar, CA</addr-line>
        <country>United States</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Alexandra H Smick <email>asmick14@gmail.com</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>29</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>12</volume>
      <elocation-id>e76471</elocation-id>
      <history>
        <date date-type="received">
          <day>23</day>
          <month>4</month>
          <year>2025</year>
        </date>
        <date date-type="rev-request">
          <day>28</day>
          <month>5</month>
          <year>2025</year>
        </date>
        <date date-type="rev-recd">
          <day>20</day>
          <month>8</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>25</day>
          <month>8</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Alexandra H Smick, Martins Ayoola, Rajeshree Rajpara, Isabel Lazo, Dana M Chase. Originally published in JMIR Cancer (https://cancer.jmir.org), 29.09.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Cancer, is properly cited. The complete bibliographic information, a link to the original publication on https://cancer.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://cancer.jmir.org/2026/1/e76471" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Patients with newly diagnosed gynecologic cancers often seek information online, but the quality of available resources may be inconsistent. Although GPT-4 may offer an alternative to traditional internet search engines, its performance remains largely under-studied in gynecologic oncology.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aimed to compare the completeness, accuracy, and reference quality of responses generated by GPT-4 with those generated by Google in clinical scenarios involving a new diagnosis of a gynecologic cancer.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>Clinical scenarios representing early- and advanced-stage endometrial, ovarian, and cervical cancers were developed by gynecologic oncologists using publicly available patient education materials. Each scenario included 4 standardized questions addressing etiology, prognosis, treatment, and treatment efficacy. GPT-4 and Google were queried for each question, with new sessions for GPT-4 and private browsing for Google to minimize bias. Responses were independently evaluated by 4 gynecologic oncology experts who were blinded to each other’s ratings. Accuracy was scored on a 6-point Likert scale; completeness and reference quality were scored on 3-point scales. Reference quality was categorized as low (commercial), medium (institutional or government), or high (peer reviewed). Optional free-text reviewer comments were collected and summarized descriptively to provide context for the quantitative findings. Descriptive statistics and Wilcoxon signed-rank tests were used for analysis.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>Across 6 clinical scenarios and 21 standardized questions (N=84 total responses), GPT-4 outperformed Google across all evaluated domains. The median accuracy score was 6.00 (IQR 5.00-6.00) for GPT-4 and 5.00 (IQR 4.00-6.00) for Google (<italic>P</italic>=.04). The median completeness score was 3.00 (IQR 3.00-3.00) for GPT-4 and 2.00 (IQR 1.00-3.00) for Google (<italic>P</italic>=.009). Reference quality was also higher for GPT-4, with a median score of 3.00 (IQR 3.00-3.00) compared to 2.00 (IQR 2.00-2.00) for Google (<italic>P</italic>=.009). Reviewer comments noted that GPT-4 provided more accurate, comprehensive, and personalized responses, while Google returned less detailed content from general consumer health websites rather than peer-reviewed sources.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>GPT-4 may serve as a reliable and high-quality resource for information in gynecologic oncology. Compared to Google, it delivered more accurate and complete content, with higher-quality references. Further research is needed to assess the readability and accessibility of GPT-4–generated content across diverse patient populations.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>artificial intelligence</kwd>
        <kwd>AI</kwd>
        <kwd>gynecologic cancer</kwd>
        <kwd>patient education as topic</kwd>
        <kwd>large language models</kwd>
        <kwd>gynecologic neoplasms</kwd>
        <kwd>health literacy</kwd>
        <kwd>cancer communication</kwd>
        <kwd>patient information materials</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>Approximately 100,000 individuals are diagnosed with gynecologic malignancy each year in the United States [<xref ref-type="bibr" rid="ref1">1</xref>]. Following a new cancer diagnosis, patients frequently seek supplemental educational materials to better understand their condition and treatment options. Traditional resources, such as those from professional societies and academic institutions, are often comprehensive but presented in generalized formats, requiring patients to interpret the content in the context of their own diagnosis and clinical stage [<xref ref-type="bibr" rid="ref2">2</xref>].</p>
      <p>AI, particularly large language models (LLMs) such as GPT-4, has gained significant attention in medicine due to its ability to generate rapid, detailed responses based on vast amounts of textual data. ChatGPT, developed by OpenAI, is powered by the GPT-4 language model and accessible to the public through the ChatGPT Plus subscription. In contrast, Google is a search engine that retrieves existing content from across the internet using proprietary algorithms [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>]. As of mid-2024, Google began integrating its own LLM, Gemini, into certain aspects of its platform. However, the standard search interface continues to primarily direct users to external web pages rather than generating original text. Previous research has explored the utility of LLMs for clinical tasks such as staging lung cancer, summarizing histopathology reports, and supporting diagnostic decision-making [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref6">6</xref>]. Although early findings are promising, concerns remain regarding the accuracy, reliability, and lack of citation transparency of AI-generated content, reinforcing the need for expert oversight.</p>
      <p>As GPT-4 becomes increasingly accessible to the public, patients are likely to turn to it, alongside traditional internet search engines, for information about their diagnosis [<xref ref-type="bibr" rid="ref2">2</xref>]. Prior studies in oncology have evaluated the quality and readability of GPT-4’s responses to patient questions, highlighting variability in content accuracy and source quality [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. Other investigations have emphasized the potential of AI to improve health literacy by simplifying medical content, yet its role in reliably supporting patient education remains unknown [<xref ref-type="bibr" rid="ref9">9</xref>]. Within gynecologic oncology specifically, research on AI-based educational tools is limited. Early studies have examined GPT-4’s ability to provide guideline-concordant recommendations for ovarian cancer diagnosis and management, but no prior work has systematically evaluated its performance in delivering high-quality informational content to newly diagnosed patients across multiple gynecologic cancers [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>].</p>
      <p>This study aimed to assess GPT-4’s effectiveness in generating patient-directed informational content in the context of newly diagnosed gynecologic cancers. We compared GPT-4 responses with Google search results across standardized clinical scenarios, focusing on 3 key domains: accuracy, completeness, and reference quality. Our findings contribute to the growing body of literature evaluating AI’s role in digital health and its potential integration into patient-centered cancer education.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Study Design</title>
        <p>This was a comparative, cross-sectional evaluation of an AI-based LLM (GPT-4; OpenAI) and a standardized internet search engine (Google; Alphabet Inc) using predefined gynecologic oncology clinical scenarios and patient-centered questions (<xref rid="figure1" ref-type="fig">Figure 1</xref>). Methods were structured and reported in accordance with the CHART (Chatbot Health Advice Reporting Tool) guideline for chatbot health advice studies [<xref ref-type="bibr" rid="ref12">12</xref>].</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Overview of the study design and evaluation process.</p>
          </caption>
          <graphic xlink:href="cancer_v12i1e76471_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Ethical Considerations</title>
        <p>Ethics approval was not required for this study because it did not involve human participants, patient data, or animal subjects. The study evaluated outputs from publicly accessible AI tools and internet search engines and therefore did not constitute human subjects research.</p>
      </sec>
      <sec>
        <title>Models</title>
        <p>ChatGPT responses were generated using the GPT-4 model via the paid ChatGPT Plus subscription (OpenAI [<xref ref-type="bibr" rid="ref13">13</xref>]). Google searches were conducted using the Chrome browser.</p>
      </sec>
      <sec>
        <title>Prompt Engineering</title>
        <p>Six clinical scenarios representing common presentations of gynecologic cancers were developed by a panel of gynecologic oncology providers. For each cancer type (ovarian, endometrial, and cervical), 7 patient-centered questions were created based on commonly asked topics in clinical practice and information available from professional society websites [<xref ref-type="bibr" rid="ref14">14</xref>], yielding a total of 21 unique questions (<xref ref-type="boxed-text" rid="box1">Textbox 1</xref>). Questions were entered verbatim into both platforms, with each question generating 1 response from GPT-4 and 1 response from Google. A new ChatGPT session was initiated prior to each question to prevent learning across questions or contextual carryover.</p>
        <p>Accuracy was assessed using a 6-point Likert scale: 1=“completely incorrect,” 2=“more incorrect than correct,” 3=“approximately equal incorrect and correct,” 4=“more correct than incorrect,” 5=“nearly all correct,” and 6=“correct.” Completeness was assessed using a 3-point Likert scale: 1=“incomplete,” addressing some aspects of the question but with significant parts missing or incomplete; 2=“adequate,” addressing all aspects of the question and providing the minimum information required to be complete; and 3=“comprehensive,” addressing all aspects of the question and providing additional information or context beyond what was expected. Reference quality was assessed using a 3-point Likert scale: 1=“bottom tier,” including any .com sources such as WebMD; 2=“middle tier,” including any .gov or .org sources such as the American Cancer Society and National Institutes of Health; and 3=“top tier,” including peer-reviewed sources or randomized controlled trials such as the National Comprehensive Cancer Network and UpToDate.</p>
        <boxed-text id="box1" position="float">
          <title>Clinical questions about diagnosis, treatment options, and survival outcomes for common gynecologic malignancies.</title>
          <list list-type="order">
            <list-item>
              <p>What causes endometrial cancer?</p>
            </list-item>
            <list-item>
              <p>How long will I live with stage IB endometrial cancer?</p>
            </list-item>
            <list-item>
              <p>What is the recommended treatment for stage IB endometrial cancer?</p>
            </list-item>
            <list-item>
              <p>How effective is first-line treatment for stage IB endometrial cancer?</p>
            </list-item>
            <list-item>
              <p>How long will I live with stage IIIC endometrial cancer?</p>
            </list-item>
            <list-item>
              <p>What is the recommended treatment for stage IIIC endometrial cancer?</p>
            </list-item>
            <list-item>
              <p>How effective is first-line treatment for stage IIIC endometrial cancer?</p>
            </list-item>
            <list-item>
              <p>What causes cervical cancer?</p>
            </list-item>
            <list-item>
              <p>How long will I live with stage IA2 cervical cancer?</p>
            </list-item>
            <list-item>
              <p>What is the recommended treatment for stage IA2 cervical cancer?</p>
            </list-item>
            <list-item>
              <p>How effective is first-line treatment for stage IA2 cervical cancer?</p>
            </list-item>
            <list-item>
              <p>How long will I live with stage IIIC1 cervical cancer?</p>
            </list-item>
            <list-item>
              <p>What is the recommended treatment for stage IIIC1 cervical cancer?</p>
            </list-item>
            <list-item>
              <p>How effective is first-line treatment for stage IIIC1 cervical cancer?</p>
            </list-item>
            <list-item>
              <p>What causes ovarian cancer?</p>
            </list-item>
            <list-item>
              <p>How long will I live with stage IC1 ovarian cancer?</p>
            </list-item>
            <list-item>
              <p>What is the recommended treatment for stage IC1 ovarian cancer?</p>
            </list-item>
            <list-item>
              <p>How effective is first-line treatment for stage IC1 ovarian cancer?</p>
            </list-item>
            <list-item>
              <p>How long will I live with stage IVB ovarian cancer?</p>
            </list-item>
            <list-item>
              <p>What is the recommended treatment for stage IVB ovarian cancer?</p>
            </list-item>
            <list-item>
              <p>How effective is first-line treatment for stage IVB ovarian cancer?</p>
            </list-item>
          </list>
        </boxed-text>
      </sec>
      <sec>
        <title>Query Strategy</title>
        <p>Google searches were conducted in private browsing mode with the browser cache cleared. For each query, the top-ranked nonsponsored web page was recorded to allow direct comparison with GPT-4’s single-response output (<xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). This approach reflects typical user behavior and aligns with prior studies demonstrating reliance on the first search results [<xref ref-type="bibr" rid="ref15">15</xref>].</p>
      </sec>
      <sec>
        <title>Performance Evaluation</title>
        <p>Responses from both platforms were evaluated for accuracy, completeness, and reference quality by 4 gynecologic oncology physicians from 2 institutions (1 academic tertiary referral center and 1 county-based safety-net hospital). Reviewers represented a range of clinical experience, with 2 physicians having &#60;5 years of clinical experience and 2 having &#62;5 years of clinical experience. Reviewers scored responses independently and were blinded to each other’s evaluations. Reviewers were also invited to provide optional free-text comments within the standardized evaluation form. These qualitative comments were summarized descriptively. Comments were reviewed in their entirety and grouped into thematic categories reflecting perceptions of accuracy, completeness, reference quality, clarity, and patient comprehensibility. A total of 76 optional comments were submitted across all reviewers. Representative comments were selected to illustrate common themes across platforms and are presented in <xref ref-type="boxed-text" rid="box2">Textbox 2</xref>. This descriptive approach aligns with conventional qualitative content analysis methods used to contextualize the quantitative findings [<xref ref-type="bibr" rid="ref16">16</xref>]. Accuracy was defined as concordance with National Comprehensive Cancer Network guidelines and current literature and scored using a 6-point Likert scale. Completeness was scored using a 3-point Likert scale based on the inclusion of major concepts relevant to each question. Reference quality was categorized as low, medium, or high and scored using a 3-point Likert scale.</p>
        <boxed-text id="box2" position="float">
          <title>Representative reviewer comments on completeness, accuracy, and reference quality for Google and GPT-4.</title>
          <p>
            <bold>Google</bold>
          </p>
          <list list-type="bullet">
            <list-item>
              <p>Technically all the information was correct, but it did not answer the specific question. They would likely need to answer this question with another source.</p>
            </list-item>
            <list-item>
              <p>The question is not answered, only treatments are mentioned. It also makes it sound like the treatments may not be that effective.</p>
            </list-item>
            <list-item>
              <p>The information looks very accurate, but there is a lot of set up with prognosis and you must scroll down to get your answer.</p>
            </list-item>
            <list-item>
              <p>Having to weed through the information to get the answer, the information is not as clear and is a little confusing.</p>
            </list-item>
            <list-item>
              <p>Some information is outdated.</p>
            </list-item>
            <list-item>
              <p>Very general information which makes it easy for a person with average literacy to understand.</p>
            </list-item>
            <list-item>
              <p>Answers to survival questions are given but there is no information about the cancer stage.</p>
            </list-item>
            <list-item>
              <p>There is no mention of treatment effectiveness or immunotherapy.</p>
            </list-item>
          </list>
          <p>
            <bold>GPT-4</bold>
          </p>
          <list list-type="bullet">
            <list-item>
              <p>Accurate information is given that is straight to the point and concise.</p>
            </list-item>
            <list-item>
              <p>Complete and thorough answers were given that were concise with high quality resources.</p>
            </list-item>
            <list-item>
              <p>ChatGPT does a nice job of putting information into context for the patient.</p>
            </list-item>
            <list-item>
              <p>ChatGPT is very comprehensive, although it does not address immunotherapy as an option.</p>
            </list-item>
            <list-item>
              <p>There is specific data of effectiveness given with very thorough data on treatment options.</p>
            </list-item>
            <list-item>
              <p>There is very good information provided on surgery and chemotherapy (including details about bevacizumab and PARPi) and also discusses palliative care.</p>
            </list-item>
            <list-item>
              <p>Information is concise and complete. It talked about what the stage meant.</p>
            </list-item>
            <list-item>
              <p>My only concern is the patient’s reading level and ability to understand the language in CGPT versus the broad, simple language used by Google.</p>
            </list-item>
          </list>
          <p>
            <bold>Overall</bold>
          </p>
          <list list-type="bullet">
            <list-item>
              <p>I thought I was going to come in biased against ChatGPT. As I looked through the answers, I think ChatGPT did much better answering questions than a random Google search with surprisingly high quality, nuanced answers.</p>
            </list-item>
            <list-item>
              <p>Google searches with answers from Mayo, COH, ACS were good for basic questions, like what is this cancer, what is the stage; although it gave you all the information on staging and you must scroll to your particular area of interest.</p>
            </list-item>
            <list-item>
              <p>ChatGPT did a better job at individualizing the answer. It also did better at prognosis, effectiveness questions. It seemed to do it with nuance and “compassion.” But it was also direct, and I think that can be helpful for physicians when it is difficult to give an accurate prognosis.</p>
            </list-item>
          </list>
        </boxed-text>
      </sec>
      <sec>
        <title>Sample Size</title>
        <p>A total of 21 standardized patient-centered questions were evaluated. Each question generated 1 response from GPT-4 and 1 response from Google, resulting in 21 responses per platform. All responses were independently evaluated by 4 gynecologic oncology physicians, yielding 84 physician evaluations per platform. The number of questions was selected to capture key informational domains, including etiology, prognosis, treatment, and treatment efficacy, while maintaining feasibility for blinded expert review. This approach is consistent with prior studies evaluating the quality of AI-generated and online health information [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref17">17</xref>-<xref ref-type="bibr" rid="ref19">19</xref>].</p>
      </sec>
      <sec>
        <title>Data Analysis</title>
        <p>Statistical analyses were conducted using JMP Pro (version 16.0.0; SAS Institute Inc). Descriptive statistics were calculated for all outcomes, and paired comparisons between platforms were performed using Wilcoxon signed-rank tests. Interrater reliability was assessed using intraclass correlation coefficients (ICCs) derived from a 2-way random-effects model with single measures and consistency (ICC[2,1]). Reliability thresholds were interpreted using established criteria [<xref ref-type="bibr" rid="ref20">20</xref>].</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <p>Accuracy was graded by 4 physicians on a Likert scale of 1 to 6, with 1=“completely incorrect” and 6=“completely correct.” The median score was 5.00 (IQR 4.00-6.00) for Google compared to 6.00 (IQR 5.00-6.00) for GPT-4 (<italic>P</italic>=.04). For Google, the distribution by accuracy score from 1 to 6 was 6% (5/84), 2.4% (2/84), 8.3% (7/84), 16.7% (14/84), 8.3% (7/84), and 58.3% (49/84), respectively. For GPT-4, the distribution from 1 to 6 was 0% (0/84), 0% (0/84), 2.4% (2/84), 10.7% (9/84), 17.8% (15/84), and 69% (58/84), respectively. A score of 5 or 6 (the most accurate) accounted for 66.6% (56/84) of responses for Google compared to 86.9% (73/84) for GPT-4 (<xref ref-type="table" rid="table1">Table 1</xref>).</p>
      <p>Completeness was then graded by 4 physicians on a Likert scale of 1 to 3, with 1=“incomplete,” addressing only some aspects of the questions, and 3=“comprehensive,” addressing all aspects of the questions. The median score was 2.00 (IQR 1.00-3.00) for Google compared to 3.00 (IQR 3.00-3.00) for GPT-4 (<italic>P</italic>=.009). For Google, the distribution by completeness score from 1 to 3 was 36.9% (31/84), 28.6% (24/84), and 34.5% (29/84), respectively. For GPT-4, the distribution from 1 to 3 was 1.2% (1/84), 17.8% (15/84), and 81% (68/84), respectively. A score of 2 or 3 (mostly complete) accounted for 63.1% (53/84) of responses for Google compared to 98.8% (83/84) for GPT-4 (<xref ref-type="table" rid="table2">Table 2</xref>).</p>
      <p>Finally, reference quality was graded by 4 physicians on a 3-point Likert scale, with 1=“bottom tier (.com)” and 3=“top tier (peer reviewed).” The median score was 2.00 (IQR 2.00-2.00) for Google compared to 3.00 (IQR 3.00-3.00) for GPT-4 (<italic>P</italic>=.009). For Google, the distribution by reference quality score from 1 to 3 was 10.7% (9/84), 77.4% (65/84), and 11.9% (10/84), respectively. For GPT-4, the distribution from 1 to 3 was 0% (0/84), 22.6% (19/84), and 77.4% (65/84), respectively. A score of 2 or 3 (highest quality references) accounted for 89.3% (75/84) of responses for Google compared to 100% (84/84) for GPT-4 (<xref ref-type="table" rid="table3">Table 3</xref>).</p>
      <p>Reviewers then provided comments on their perceptions of the accuracy, completeness, and reference quality of GPT-4 compared to Google. A total of 76 optional comments were submitted across all reviewers. Representative comments illustrating the major themes identified across reviewers are presented in <xref ref-type="boxed-text" rid="box2">Textbox 2</xref>. Overall, reviewers noted that Google source responses had clear and precise language but often did not completely answer the question and did not provide enough information on prognosis or treatment options. For GPT-4, reviewers generally commented that answers were very complete and comprehensive, especially regarding disease stage and prognosis, but might be more challenging for patients to understand.</p>
      <p>To assess the consistency of reviewer scoring, we calculated interrater reliability using ICCs based on a 2-way random-effects model evaluating consistency. Completeness scores for Google demonstrated moderate consistency (ICC 0.595, 95% CI 0.389-0.782), while all other domains showed poor consistency (ICC&#60;0.50), except for GPT-4 reference quality, which demonstrated excellent interrater reliability (ICC 0.935, 95% CI 0.879-0.970). Full results are shown in <xref ref-type="table" rid="table4">Table 4</xref>.</p>
      <table-wrap position="float" id="table1">
        <label>Table 1</label>
        <caption>
          <p>Comparison of accuracy scores for clinical scenarios between GPT-4 and Google.</p>
        </caption>
        <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
          <col width="370"/>
          <col width="290"/>
          <col width="340"/>
          <thead>
            <tr valign="top">
              <td>Measure</td>
              <td>Google (n=84)</td>
              <td>GPT-4 (n=84)</td>
            </tr>
          </thead>
          <tbody>
            <tr valign="top">
              <td>Accuracy score 1, n (%)</td>
              <td>5 (6)</td>
              <td>0 (0)</td>
            </tr>
            <tr valign="top">
              <td>Accuracy score 2, n (%)</td>
              <td>2 (2.4)</td>
              <td>0 (0)</td>
            </tr>
            <tr valign="top">
              <td>Accuracy score 3, n (%)</td>
              <td>7 (8.3)</td>
              <td>2 (2.4)</td>
            </tr>
            <tr valign="top">
              <td>Accuracy score 4, n (%)</td>
              <td>14 (16.7)</td>
              <td>9 (10.7)</td>
            </tr>
            <tr valign="top">
              <td>Accuracy score 5, n (%)</td>
              <td>7 (8.3)</td>
              <td>15 (17.8)</td>
            </tr>
            <tr valign="top">
              <td>Accuracy score 6, n (%)</td>
              <td>49 (58.3)</td>
              <td>58 (69)</td>
            </tr>
            <tr valign="top">
              <td>Mean (SD)<sup>a</sup></td>
              <td>4.94 (1.51)</td>
              <td>5.54 (0.78)</td>
            </tr>
            <tr valign="top">
              <td>Median (IQR)<sup>a</sup></td>
              <td>6.00 (4.00-6.00)</td>
              <td>6.00 (5.00-6.00)</td>
            </tr>
          </tbody>
        </table>
        <table-wrap-foot>
          <fn id="table1fn1">
            <p><sup>a</sup><italic>P</italic>=.04.</p>
          </fn>
        </table-wrap-foot>
      </table-wrap>
      <table-wrap position="float" id="table2">
        <label>Table 2</label>
        <caption>
          <p>Comparison of completeness scores for clinical scenarios between GPT-4 and Google.</p>
        </caption>
        <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
          <col width="360"/>
          <col width="330"/>
          <col width="310"/>
          <thead>
            <tr valign="top">
              <td>Measure</td>
              <td>Google (n=84)</td>
              <td>GPT-4 (n=84)</td>
            </tr>
          </thead>
          <tbody>
            <tr valign="top">
              <td>Completeness score 1, n (%)</td>
              <td>31 (36.9)</td>
              <td>1 (1.2)</td>
            </tr>
            <tr valign="top">
              <td>Completeness score 2, n (%)</td>
              <td>24 (28.6)</td>
              <td>15 (17.8)</td>
            </tr>
            <tr valign="top">
              <td>Completeness score 3, n (%)</td>
              <td>29 (34.5)</td>
              <td>68 (81)</td>
            </tr>
            <tr valign="top">
              <td>Mean (SD)<sup>a</sup></td>
              <td>1.98 (0.85)</td>
              <td>2.80 (0.43)</td>
            </tr>
            <tr valign="top">
              <td>Median (IQR)<sup>a</sup></td>
              <td>2.00 (1.00-3.00)</td>
              <td>3.00 (3.00-3.00)</td>
            </tr>
          </tbody>
        </table>
        <table-wrap-foot>
          <fn id="table2fn1">
            <p><sup>a</sup><italic>P</italic>=.009.</p>
          </fn>
        </table-wrap-foot>
      </table-wrap>
      <table-wrap position="float" id="table3">
        <label>Table 3</label>
        <caption>
          <p>Comparison of reference quality scores for clinical scenarios between GPT-4 and Google.</p>
        </caption>
        <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
          <col width="370"/>
          <col width="320"/>
          <col width="310"/>
          <thead>
            <tr valign="top">
              <td>Measure</td>
              <td>Google (n=84)</td>
              <td>GPT-4 (n=84)</td>
            </tr>
          </thead>
          <tbody>
            <tr valign="top">
              <td>Reference quality score 1, n (%)</td>
              <td>9 (10.7)</td>
              <td>0 (0)</td>
            </tr>
            <tr valign="top">
              <td>Reference quality score 2, n (%)</td>
              <td>65 (77.4)</td>
              <td>19 (22.6)</td>
            </tr>
            <tr valign="top">
              <td>Reference quality score 3, n (%)</td>
              <td>10 (11.9)</td>
              <td>65 (77.4)</td>
            </tr>
            <tr valign="top">
              <td>Mean (SD)<sup>a</sup></td>
              <td>2.01 (0.48)</td>
              <td>2.77 (0.42)</td>
            </tr>
            <tr valign="top">
              <td>Median (IQR)<sup>a</sup></td>
              <td>2.00 (2.00-2.00)</td>
              <td>3.00 (3.00-3.00)</td>
            </tr>
          </tbody>
        </table>
        <table-wrap-foot>
          <fn id="table3fn1">
            <p><sup>a</sup><italic>P</italic>=.009.</p>
          </fn>
        </table-wrap-foot>
      </table-wrap>
      <table-wrap position="float" id="table4">
        <label>Table 4</label>
        <caption>
          <p>Interrater reliability by source and scoring domain.</p>
        </caption>
        <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
          <col width="30"/>
          <col width="350"/>
          <col width="390"/>
          <col width="230"/>
          <thead>
            <tr valign="top">
              <td colspan="2">Source and scoring domains</td>
              <td>Interclass correlation<sup>a</sup> (95% CI)</td>
              <td>Interpretation<sup>b</sup></td>
            </tr>
          </thead>
          <tbody>
            <tr valign="top">
              <td colspan="4">
                <bold>Google</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>Accuracy</td>
              <td>0.075 (−0.087 to 0.327)</td>
              <td>Poor</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>Completeness</td>
              <td>0.595 (0.389 to 0.782)</td>
              <td>Moderate</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>Reference quality</td>
              <td>0.016 (−0.128 to 0.255)</td>
              <td>Poor</td>
            </tr>
            <tr valign="top">
              <td colspan="4">
                <bold>GPT-4</bold>
              </td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>Accuracy</td>
              <td>0.111 (−0.061 to 0369)</td>
              <td>Poor</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>Completeness</td>
              <td>0 (−0.138 to 0.234)</td>
              <td>Poor</td>
            </tr>
            <tr valign="top">
              <td>
                <break/>
              </td>
              <td>Reference quality</td>
              <td>0.935 (0.879 to 0.970)</td>
              <td>Excellent</td>
            </tr>
          </tbody>
        </table>
        <table-wrap-foot>
          <fn id="table4fn1">
            <p><sup>a</sup>Intraclass correlation coefficients were calculated using a 2-way random-effects model with single measures (ICC[2,1]).</p>
          </fn>
          <fn id="table4fn2">
            <p><sup>b</sup>Poor&#60;0.5, moderate=0.5-0.75, good=0.75-0.90, and excellent&#62;0.90.</p>
          </fn>
        </table-wrap-foot>
      </table-wrap>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <sec>
        <title>Principal Results</title>
        <p>This study found that GPT-4 performed better than Google sources when providing high-quality responses to clinical scenarios involving a new diagnosis of gynecologic cancer. Physician reviewers rated GPT-4 significantly higher than Google in terms of accuracy, completeness, and reference quality. Reviewers were surprised by the quality of the GPT-4 answers and references and saw the potential for GPT-4 to better address patient-specific questions regarding diagnosis, prognosis, and treatment options with both precise and comprehensive information. The most common shortcoming identified in the Google-sourced content was the provision of general information that was not specific to the patient’s cancer diagnosis, stage, or treatment options and therefore was difficult to individually interpret.</p>
      </sec>
      <sec>
        <title>Comparison With Prior Work</title>
        <p>Although a growing number of studies have evaluated GPT-4 for cancer-related patient education, relatively few have focused specifically on gynecologic oncology. Initial research on AI in medicine focused on its diagnostic role and potential for clinical decision-making. For example, in otorhinolaryngology, GPT-4 was compared to UpToDate in terms of usefulness and reliability in diagnosing certain common clinical scenarios. UpToDate was found to be more accurate than GPT-4 (<italic>P</italic>=.009), and the authors concluded that improvements were needed in the usefulness and reliability of GPT-4’s evidence-based knowledge [<xref ref-type="bibr" rid="ref21">21</xref>]. Similarly, in various clinical cases of thyroid cancer, GPT-4 was found to be reasonably accurate (approximately 76.7%) when providing information about thyroid cancer, but recommendations about evaluation, treatment, and follow-up were deemed general and inadequate [<xref ref-type="bibr" rid="ref22">22</xref>].</p>
        <p>More recently, studies in gynecologic oncology have demonstrated that GPT-4 generally provides guideline-concordant recommendations for gynecologic cancer diagnosis and management, while emphasizing the continued importance of physician oversight for complex clinical decision-making [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref23">23</xref>]. Chou et al [<xref ref-type="bibr" rid="ref17">17</xref>] further reported that GPT-4 generated patient-directed responses regarding ovarian cancer that were comparable to those provided by gynecologic oncology clinicians, supporting its potential role as a patient education tool.</p>
        <p>As AI is increasingly explored as a tool for patient-facing communication, several prior studies have evaluated the use of GPT-4 to generate patient information. One study used various LLM platforms to generate patient education materials for both common and rare dermatologic conditions and found that these platforms were able to create accessible, understandable content at an appropriate reading level [<xref ref-type="bibr" rid="ref18">18</xref>]. Another study focused on the ability of GPT-4 to provide patients with information about screening mammography but documented concerns about the understandability and actionability of the answers, specifically from a health literacy standpoint [<xref ref-type="bibr" rid="ref19">19</xref>]. Although AI seems to offer an opportunity to enhance patients’ access to health information, there continues to be concern about the possibility of misinformation and the quality of references.</p>
        <p>Our findings build upon this growing body of literature. Compared to top-ranked Google search results, GPT-4 consistently outperformed in terms of accuracy, completeness, and reference quality. These findings are consistent with prior studies demonstrating that GPT-4 generally provides accurate, comprehensive, and guideline-concordant information across a variety of cancer-related clinical scenarios [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. Consistent with the findings of Chou et al [<xref ref-type="bibr" rid="ref17">17</xref>], we found that GPT-4 generated comprehensive, patient-directed responses that were highly rated by gynecologic oncology physicians. Similarly, our findings align with those of Piazza et al [<xref ref-type="bibr" rid="ref10">10</xref>] and Finch et al [<xref ref-type="bibr" rid="ref11">11</xref>], who reported strong agreement between ChatGPT-generated recommendations and established ovarian cancer guidelines. However, although previous studies primarily compared ChatGPT with clinical guidelines or clinician responses, our study directly compared GPT-4 with a conventional Google search strategy, reflecting a common approach used by patients when seeking information after a new cancer diagnosis. Reviewers noted that GPT-4 responses were generally more comprehensive and better aligned with the clinical context, often including relevant treatment information and prognostic details that were lacking in standard search results. Although previous concerns have been raised about the citation quality of previous GPT-4 models, our findings suggest that, in this study, the reference quality was significantly higher than that of Google search results [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. Overall, these results support the potential of GPT-4 to generate personalized, high-quality responses that may enhance patient understanding, support shared decision-making, and improve access to reliable health information.</p>
      </sec>
      <sec>
        <title>Strengths and Limitations</title>
        <p>Strengths of this study include the novel design used to investigate the role of AI in generating informational content related to gynecologic malignancies. By querying GPT-4 and Google with multiple patient scenarios, results were more generalizable to patients with different types and stages of gynecologic cancers. The use of gynecologic oncology physicians, who were blinded to each other’s assessments, enabled the clinical evaluation of response quality by experts who routinely guide patients through diagnosis and treatment. Evaluation of accuracy, completeness, and reference quality offered a comprehensive assessment of the content delivered by each platform. Importantly, we also assessed interrater reliability, which is often omitted in similar studies. Although agreement varied by domain, the inclusion of this analysis strengthens the rigor of our evaluation.</p>
        <p>Some limitations of this study include a relatively small number of reviewers, a subjective grading scale, and no direct assessment of the readability of the sources used. Although reviewers were blinded to each other’s ratings, they were aware of the source of each response (GPT-4 or Google), which may have introduced bias. Additionally, all 4 reviewers were from only 2 institutions, which raises the possibility that assessments may have been influenced by local practice patterns or institutional standards, potentially limiting generalizability. Interrater reliability was variable across domains, which likely reflects the inherently subjective nature of evaluating informational content, particularly when responses were largely accurate and complete, resulting in clustering of scores at the higher end of the grading scales. In this context, small differences in individual scoring behavior may disproportionately affect interrater reliability metrics despite overall agreement in comparative performance between platforms. Another important consideration is that the phrasing of questions may influence the quality and clarity of responses generated by LLMs. In this study, the scenarios were designed to reflect the types of questions a well-informed patient might ask after receiving a new cancer diagnosis. In real-world settings, however, patients may not know what questions to ask, may lack the background to phrase them effectively, or may not have the clinical knowledge needed to interpret detailed responses. GPT-4 does not tailor responses based on a user’s health literacy level, and this study did not formally assess the readability or accessibility of the responses. This represents an important limitation, as prior research has shown that the effectiveness of patient education materials depends not only on content accuracy but also on whether they are comprehensible and actionable. Additionally, by evaluating only the top-ranked Google result for each query, the study may not fully capture the breadth of information users might access through additional searches.</p>
        <p>Since completing this study, Google has expanded the integration of LLM-generated summaries within its search platform. This underscores the rapidly evolving landscape of online health information and highlights the need for continued evaluation of AI-assisted search tools. Although such integrations may narrow performance gaps between traditional search engines and stand-alone LLMs, our findings remain relevant by illustrating the potential advantages of conversational, context-aware responses over static web page retrieval. Future studies should directly compare AI-generated summaries embedded within search engines with stand-alone LLMs, with a focus on reference transparency, patient personalization, and health literacy.</p>
      </sec>
      <sec>
        <title>Conclusions</title>
        <p>The results of this study suggest that GPT-4 may be a useful tool for delivering accurate, comprehensive, and well-referenced informational content in gynecologic oncology when compared with top-ranked Google search results within the context of standardized clinical scenarios. However, these findings should be interpreted cautiously given several important methodological limitations, including incomplete blinding of reviewers to response source, variable interrater reliability across evaluation domains, and the restriction of the Google comparator to a single top-ranked result per query. Accordingly, this study should be viewed as a preliminary, hypothesis-generating evaluation rather than a definitive assessment of comparative performance. Although LLMs such as GPT-4 show promise in generating context-aware responses tailored to specific clinical scenarios, further research is needed to clarify their appropriate role in supporting patient education and health care professional–patient communication in gynecologic oncology.</p>
      </sec>
    </sec>
  </body>
  <back>
    <app-group>
      <supplementary-material id="app1">
        <label>Multimedia Appendix 1</label>
        <p>Gynecologic oncology case questions, source materials, and physician grading rubrics used to evaluate Google and GPT-4 responses.</p>
        <media xlink:href="cancer_v12i1e76471_app1.docx" xlink:title="DOCX File , 44 KB"/>
      </supplementary-material>
    </app-group>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">CHART</term>
          <def>
            <p>Chatbot Health Advice Reporting Tool</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">ICC</term>
          <def>
            <p>intraclass correlation coefficient</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">LLM</term>
          <def>
            <p>large language model</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The authors declare that generative AI (GPT-4; OpenAI) was used as described in the Methods section.</p>
    </ack>
    <notes>
      <sec>
        <title>Funding</title>
        <p>The authors declare that no financial support was received for this work.</p>
      </sec>
    </notes>
    <fn-group>
      <fn fn-type="con">
        <p>Conceptualization: AHS, IL, DMC</p>
        <p>Data collection: AHS, MA, IL, RR, DMC</p>
        <p>Statistical analysis: AHS, MA</p>
        <p>Writing—original manuscript: AHS, RR</p>
        <p>Writing—review and editing: MA, IL, DMC</p>
      </fn>
      <fn fn-type="conflict">
        <p>DMC has served on speakers’ bureaus for Corcept, AstraZeneca, GSK, AbbVie, and Pfizer; has served as a consultant or advisor for Genmab, Corcept, AstraZeneca, GSK, AbbVie, and Pfizer; and has received research funding from GSK and Merck.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="web">
          <article-title>SEER*Stat databases: SEER November 2023 submission</article-title>
          <source>National Cancer Institute</source>
          <year>2023</year>
          <access-date>2025-07-01</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.seer.cancer.gov">https://www.seer.cancer.gov</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bhattad</surname>
              <given-names>PB</given-names>
            </name>
            <name name-style="western">
              <surname>Pacifico</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>Empowering patients: promoting patient education and health literacy</article-title>
          <source>Cureus</source>
          <year>2022</year>
          <month>07</month>
          <day>27</day>
          <volume>14</volume>
          <issue>7</issue>
          <fpage>e27336</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/36043002"/>
          </comment>
          <pub-id pub-id-type="doi">10.7759/cureus.27336</pub-id>
          <pub-id pub-id-type="medline">36043002</pub-id>
          <pub-id pub-id-type="pmcid">PMC9411825</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Iannantuono</surname>
              <given-names>GM</given-names>
            </name>
            <name name-style="western">
              <surname>Bracken-Clarke</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Floudas</surname>
              <given-names>CS</given-names>
            </name>
            <name name-style="western">
              <surname>Roselli</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gulley</surname>
              <given-names>JL</given-names>
            </name>
            <name name-style="western">
              <surname>Karzai</surname>
              <given-names>F</given-names>
            </name>
          </person-group>
          <article-title>Applications of large language models in cancer care: current evidence and future perspectives</article-title>
          <source>Front Oncol</source>
          <year>2023</year>
          <month>9</month>
          <day>4</day>
          <volume>13</volume>
          <fpage>1268915</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/37731643"/>
          </comment>
          <pub-id pub-id-type="doi">10.3389/fonc.2023.1268915</pub-id>
          <pub-id pub-id-type="medline">37731643</pub-id>
          <pub-id pub-id-type="pmcid">PMC10507617</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Haug</surname>
              <given-names>CJ</given-names>
            </name>
            <name name-style="western">
              <surname>Drazen</surname>
              <given-names>JM</given-names>
            </name>
          </person-group>
          <article-title>Artificial intelligence and machine learning in clinical medicine, 2023</article-title>
          <source>N Engl J Med</source>
          <year>2023</year>
          <month>03</month>
          <day>30</day>
          <volume>388</volume>
          <issue>13</issue>
          <fpage>1201</fpage>
          <lpage>8</lpage>
          <pub-id pub-id-type="doi">10.1056/NEJMra2302038</pub-id>
          <pub-id pub-id-type="medline">36988595</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tozuka</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Johno</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Amakawa</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sato</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Muto</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Seki</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Komaba</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Onishi</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Application of NotebookLM, a large language model with retrieval-augmented generation, for lung cancer staging</article-title>
          <source>Jpn J Radiol</source>
          <year>2025</year>
          <month>04</month>
          <volume>43</volume>
          <issue>4</issue>
          <fpage>706</fpage>
          <lpage>12</lpage>
          <pub-id pub-id-type="doi">10.1007/s11604-024-01705-1</pub-id>
          <pub-id pub-id-type="medline">39585559</pub-id>
          <pub-id pub-id-type="pii">10.1007/s11604-024-01705-1</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hewitt</surname>
              <given-names>KJ</given-names>
            </name>
            <name name-style="western">
              <surname>Wiest</surname>
              <given-names>IC</given-names>
            </name>
            <name name-style="western">
              <surname>Carrero</surname>
              <given-names>ZI</given-names>
            </name>
            <name name-style="western">
              <surname>Bejan</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Millner</surname>
              <given-names>TO</given-names>
            </name>
            <name name-style="western">
              <surname>Brandner</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kather</surname>
              <given-names>JN</given-names>
            </name>
          </person-group>
          <article-title>Large language models as a diagnostic support tool in neuropathology</article-title>
          <source>J Pathol Clin Res</source>
          <year>2024</year>
          <month>11</month>
          <volume>10</volume>
          <issue>6</issue>
          <fpage>e70009</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://onlinelibrary.wiley.com/doi/10.1002/2056-4538.70009"/>
          </comment>
          <pub-id pub-id-type="doi">10.1002/2056-4538.70009</pub-id>
          <pub-id pub-id-type="medline">39505569</pub-id>
          <pub-id pub-id-type="pmcid">PMC11540532</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Anderson</surname>
              <given-names>PB</given-names>
            </name>
            <name name-style="western">
              <surname>Wanken</surname>
              <given-names>ZJ</given-names>
            </name>
            <name name-style="western">
              <surname>Perri</surname>
              <given-names>JL</given-names>
            </name>
            <name name-style="western">
              <surname>Columbo</surname>
              <given-names>JA</given-names>
            </name>
            <name name-style="western">
              <surname>Kang</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Spangler</surname>
              <given-names>EL</given-names>
            </name>
            <name name-style="western">
              <surname>Newhall</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Brooke</surname>
              <given-names>BS</given-names>
            </name>
            <name name-style="western">
              <surname>Dosluoglu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>ES</given-names>
            </name>
            <name name-style="western">
              <surname>Raffetto</surname>
              <given-names>JD</given-names>
            </name>
            <name name-style="western">
              <surname>Henke</surname>
              <given-names>PK</given-names>
            </name>
            <name name-style="western">
              <surname>Tang</surname>
              <given-names>GL</given-names>
            </name>
            <name name-style="western">
              <surname>Mureebe</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Kougias</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Johanning</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Arya</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Scali</surname>
              <given-names>ST</given-names>
            </name>
            <name name-style="western">
              <surname>Stone</surname>
              <given-names>DH</given-names>
            </name>
            <name name-style="western">
              <surname>Suckow</surname>
              <given-names>BD</given-names>
            </name>
            <name name-style="western">
              <surname>Orion</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Halpern</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>O'Connell</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Inhat</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Nelson</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Tzeng</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Barry</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Sirovich</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Goodney</surname>
              <given-names>PP</given-names>
            </name>
          </person-group>
          <article-title>Patient information sources when facing repair of abdominal aortic aneurysm</article-title>
          <source>J Vasc Surg</source>
          <year>2020</year>
          <month>02</month>
          <volume>71</volume>
          <issue>2</issue>
          <fpage>497</fpage>
          <lpage>504</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S0741-5214(19)31127-9"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jvs.2019.04.460</pub-id>
          <pub-id pub-id-type="medline">31353272</pub-id>
          <pub-id pub-id-type="pii">S0741-5214(19)31127-9</pub-id>
          <pub-id pub-id-type="pmcid">PMC10767985</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Łaszkiewicz</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Krajewski</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Tomczak</surname>
              <given-names>W</given-names>
            </name>
            <name name-style="western">
              <surname>Chorbińska</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Nowak</surname>
              <given-names>Ł</given-names>
            </name>
            <name name-style="western">
              <surname>Chełmoński</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Krajewski</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Sójka</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Małkiewicz</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Szydełko</surname>
              <given-names>T</given-names>
            </name>
          </person-group>
          <article-title>Performance of ChatGPT in providing patient information about upper tract urothelial carcinoma</article-title>
          <source>Contemp Oncol (Pozn)</source>
          <year>2024</year>
          <volume>28</volume>
          <issue>2</issue>
          <fpage>172</fpage>
          <lpage>81</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.5114/wo.2024.141567"/>
          </comment>
          <pub-id pub-id-type="doi">10.5114/wo.2024.141567</pub-id>
          <pub-id pub-id-type="medline">39421706</pub-id>
          <pub-id pub-id-type="pii">54492</pub-id>
          <pub-id pub-id-type="pmcid">PMC11480910</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gupta</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Gupta</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Ho</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Wood</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Guleria</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Virostko</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Can generative AI improve the readability of patient education materials at a radiology practice?</article-title>
          <source>Clin Radiol</source>
          <year>2024</year>
          <month>11</month>
          <volume>79</volume>
          <issue>11</issue>
          <fpage>e1366</fpage>
          <lpage>71</lpage>
          <pub-id pub-id-type="doi">10.1016/j.crad.2024.08.019</pub-id>
          <pub-id pub-id-type="medline">39266371</pub-id>
          <pub-id pub-id-type="pii">S0009-9260(24)00431-8</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Piazza</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Martorana</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Curaba</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sambataro</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Valerio</surname>
              <given-names>MR</given-names>
            </name>
            <name name-style="western">
              <surname>Firenze</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Pecorino</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Scollo</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Chiantera</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Scibilia</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Vigneri</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Gebbia</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Scandurra</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>The consistency and quality of ChatGPT responses compared to clinical guidelines for ovarian cancer: a Delphi approach</article-title>
          <source>Curr Oncol</source>
          <year>2024</year>
          <month>05</month>
          <day>14</day>
          <volume>31</volume>
          <issue>5</issue>
          <fpage>2796</fpage>
          <lpage>804</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=curroncol31050212"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/curroncol31050212</pub-id>
          <pub-id pub-id-type="medline">38785493</pub-id>
          <pub-id pub-id-type="pii">curroncol31050212</pub-id>
          <pub-id pub-id-type="pmcid">PMC11119344</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Finch</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Broach</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Feinberg</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Al-Niaimi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Abu-Rustum</surname>
              <given-names>NR</given-names>
            </name>
            <name name-style="western">
              <surname>Zhou</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Iasonos</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Chi</surname>
              <given-names>DS</given-names>
            </name>
          </person-group>
          <article-title>ChatGPT compared to national guidelines for management of ovarian cancer: did ChatGPT get it right? - A Memorial Sloan Kettering Cancer Center team ovary study</article-title>
          <source>Gynecol Oncol</source>
          <year>2024</year>
          <month>10</month>
          <volume>189</volume>
          <fpage>75</fpage>
          <lpage>9</lpage>
          <pub-id pub-id-type="doi">10.1016/j.ygyno.2024.07.007</pub-id>
          <pub-id pub-id-type="medline">39042956</pub-id>
          <pub-id pub-id-type="pii">S0090-8258(24)00362-7</pub-id>
          <pub-id pub-id-type="pmcid">PMC11402584</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <collab>CHART Collaborative</collab>
          </person-group>
          <article-title>Reporting guideline for chatbot health advice studies: Chatbot Assessment Reporting Tool (CHART) statement</article-title>
          <source>Ann Fam Med</source>
          <year>2025</year>
          <month>09</month>
          <day>22</day>
          <volume>23</volume>
          <issue>5</issue>
          <fpage>389</fpage>
          <lpage>98</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://www.annfammed.org/cgi/pmidlookup?view=long&#38;pmid=40750305"/>
          </comment>
          <pub-id pub-id-type="doi">10.1370/afm.250386</pub-id>
          <pub-id pub-id-type="medline">40750305</pub-id>
          <pub-id pub-id-type="pii">afm.250386</pub-id>
          <pub-id pub-id-type="pmcid">PMC12459699</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="web">
          <source>ChatGPT</source>
          <access-date>2026-09-09</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://chatgpt.com/">https://chatgpt.com/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="web">
          <article-title>Educational materials</article-title>
          <source>Foundation for Women’s Cancer</source>
          <access-date>2025-07-01</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://foundationforwomenscancer.org/resources/educational-materials/">https://foundationforwomenscancer.org/resources/educational-materials/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Urman</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Makhortykh</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>You are how (and where) you search? Comparative analysis of web search behavior using web tracking data</article-title>
          <source>J Comput Soc Sci</source>
          <year>2023</year>
          <month>05</month>
          <day>03</day>
          <volume>6</volume>
          <issue>2</issue>
          <fpage>1</fpage>
          <lpage>16</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://boris-portal.unibe.ch/handle/20.500.12422/168173"/>
          </comment>
          <pub-id pub-id-type="doi">10.1007/s42001-023-00208-9</pub-id>
          <pub-id pub-id-type="medline">37363807</pub-id>
          <pub-id pub-id-type="pii">208</pub-id>
          <pub-id pub-id-type="pmcid">PMC10155157</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hsieh</surname>
              <given-names>HF</given-names>
            </name>
            <name name-style="western">
              <surname>Shannon</surname>
              <given-names>SE</given-names>
            </name>
          </person-group>
          <article-title>Three approaches to qualitative content analysis</article-title>
          <source>Qual Health Res</source>
          <year>2005</year>
          <month>11</month>
          <volume>15</volume>
          <issue>9</issue>
          <fpage>1277</fpage>
          <lpage>88</lpage>
          <pub-id pub-id-type="doi">10.1177/1049732305276687</pub-id>
          <pub-id pub-id-type="medline">16204405</pub-id>
          <pub-id pub-id-type="pii">15/9/1277</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chou</surname>
              <given-names>HH</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>YH</given-names>
            </name>
            <name name-style="western">
              <surname>Lin</surname>
              <given-names>CT</given-names>
            </name>
            <name name-style="western">
              <surname>Chang</surname>
              <given-names>HT</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>AC</given-names>
            </name>
            <name name-style="western">
              <surname>Tsai</surname>
              <given-names>JL</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>HW</given-names>
            </name>
            <name name-style="western">
              <surname>Hsu</surname>
              <given-names>CC</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>SY</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>JT</given-names>
            </name>
          </person-group>
          <article-title>AI-driven patient support: evaluating the effectiveness of ChatGPT-4 in addressing queries about ovarian cancer compared with healthcare professionals in gynecologic oncology</article-title>
          <source>Support Care Cancer</source>
          <year>2025</year>
          <month>04</month>
          <day>01</day>
          <volume>33</volume>
          <issue>4</issue>
          <fpage>337</fpage>
          <pub-id pub-id-type="doi">10.1007/s00520-025-09389-7</pub-id>
          <pub-id pub-id-type="medline">40167802</pub-id>
          <pub-id pub-id-type="pii">10.1007/s00520-025-09389-7</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Lambert</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Choo</surname>
              <given-names>ZY</given-names>
            </name>
            <name name-style="western">
              <surname>Gradwohl</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Schroedl</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Ruiz De Luzuriaga</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Assessing the application of large language models in generating dermatologic patient education materials according to reading level: qualitative study</article-title>
          <source>JMIR Dermatol</source>
          <year>2024</year>
          <month>05</month>
          <day>16</day>
          <volume>7</volume>
          <fpage>e55898</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://derma.jmir.org/2024//e55898/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/55898</pub-id>
          <pub-id pub-id-type="medline">38754096</pub-id>
          <pub-id pub-id-type="pii">v7i1e55898</pub-id>
          <pub-id pub-id-type="pmcid">PMC11140271</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Spuur</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Currie</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Al-Mousa</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Pape</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Suitability of ChatGPT as a source of patient information for screening mammography</article-title>
          <source>Health Promot Pract</source>
          <year>2025</year>
          <month>07</month>
          <volume>26</volume>
          <issue>4</issue>
          <fpage>746</fpage>
          <lpage>62</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://journals.sagepub.com/doi/10.1177/15248399241285060?url_ver=Z39.88-2003&#38;rfr_id=ori:rid:crossref.org&#38;rfr_dat=cr_pub  0pubmed"/>
          </comment>
          <pub-id pub-id-type="doi">10.1177/15248399241285060</pub-id>
          <pub-id pub-id-type="medline">39392690</pub-id>
          <pub-id pub-id-type="pmcid">PMC12149468</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Koo</surname>
              <given-names>TK</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>MY</given-names>
            </name>
          </person-group>
          <article-title>A guideline of selecting and reporting intraclass correlation coefficients for reliability research</article-title>
          <source>J Chiropr Med</source>
          <year>2016</year>
          <month>06</month>
          <volume>15</volume>
          <issue>2</issue>
          <fpage>155</fpage>
          <lpage>63</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/27330520"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.jcm.2016.02.012</pub-id>
          <pub-id pub-id-type="medline">27330520</pub-id>
          <pub-id pub-id-type="pii">S1556-3707(16)00015-8</pub-id>
          <pub-id pub-id-type="pmcid">PMC4913118</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Karimov</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Allahverdiyev</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Agayarov</surname>
              <given-names>OY</given-names>
            </name>
            <name name-style="western">
              <surname>Demir</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Almuradova</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>ChatGPT vs UpToDate: comparative study of usefulness and reliability of Chatbot in common clinical presentations of otorhinolaryngology-head and neck surgery</article-title>
          <source>Eur Arch Otorhinolaryngol</source>
          <year>2024</year>
          <month>04</month>
          <volume>281</volume>
          <issue>4</issue>
          <fpage>2145</fpage>
          <lpage>51</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/38217726"/>
          </comment>
          <pub-id pub-id-type="doi">10.1007/s00405-023-08423-w</pub-id>
          <pub-id pub-id-type="medline">38217726</pub-id>
          <pub-id pub-id-type="pii">10.1007/s00405-023-08423-w</pub-id>
          <pub-id pub-id-type="pmcid">PMC10942922</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Cavnar Helvaci</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Hepsen</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Candemir</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Boz</surname>
              <given-names>O</given-names>
            </name>
            <name name-style="western">
              <surname>Durantas</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Houssein</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Cakal</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Assessing the accuracy and reliability of ChatGPT's medical responses about thyroid cancer</article-title>
          <source>Int J Med Inform</source>
          <year>2024</year>
          <month>11</month>
          <volume>191</volume>
          <fpage>105593</fpage>
          <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2024.105593</pub-id>
          <pub-id pub-id-type="medline">39151245</pub-id>
          <pub-id pub-id-type="pii">S1386-5056(24)00256-9</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Reicher</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Lutsker</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Michaan</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Grisaru</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Laskov</surname>
              <given-names>I</given-names>
            </name>
          </person-group>
          <article-title>Exploring the role of artificial intelligence, large language models: comparing patient-focused information and clinical decision support capabilities to the gynecologic oncology guidelines</article-title>
          <source>Int J Gynaecol Obstet</source>
          <year>2025</year>
          <month>02</month>
          <volume>168</volume>
          <issue>2</issue>
          <fpage>419</fpage>
          <lpage>27</lpage>
          <pub-id pub-id-type="doi">10.1002/ijgo.15869</pub-id>
          <pub-id pub-id-type="medline">39161265</pub-id>
          <pub-id pub-id-type="pmcid">PMC11726133</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
