<?xml version="1.0" encoding="iso-8859-1" standalone="no"?>
<!DOCTYPE GmsArticle SYSTEM "http://www.egms.de/dtd/2.0.34/GmsArticle.dtd">
<GmsArticle xmlns:xlink="http://www.w3.org/1999/xlink">
  <MetaData>
    <Identifier>26gma105</Identifier>
    <IdentifierDoi>10.3205/26gma105</IdentifierDoi>
    <IdentifierUrn>urn:nbn:de:0183-26gma1058</IdentifierUrn>
    <ArticleType>Meeting Abstract</ArticleType>
    <TitleGroup>
      <Title language="en">Large language model-based automated scoring of clinical summary statements in medical training</Title>
    </TitleGroup>
    <CreatorList>
      <Creator>
        <PersonNames>
          <Lastname>Stadler</Lastname>
          <LastnameHeading>Stadler</LastnameHeading>
          <Firstname>Patrick</Firstname>
          <Initials>P</Initials>
        </PersonNames>
        <Address>
          <Affiliation>Charit&#233; &#8211; Universit&#228;tsmedizin Berlin, Dieter Scheffner Center for Medical Education and Educational Research, Berlin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="yes">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Bucichowski</Lastname>
          <LastnameHeading>Bucichowski</LastnameHeading>
          <Firstname>Piotr</Firstname>
          <Initials>P</Initials>
        </PersonNames>
        <Address>
          <Affiliation>Charit&#233; &#8211; Universit&#228;tsmedizin Berlin, Dieter Scheffner Center for Medical Education and Educational Research, Berlin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Holzhausen</Lastname>
          <LastnameHeading>Holzhausen</LastnameHeading>
          <Firstname>Ylva</Firstname>
          <Initials>Y</Initials>
        </PersonNames>
        <Address>
          <Affiliation>Charit&#233; &#8211; Universit&#228;tsmedizin Berlin, Dieter Scheffner Center for Medical Education and Educational Research, Berlin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Hege</Lastname>
          <LastnameHeading>Hege</LastnameHeading>
          <Firstname>Inga</Firstname>
          <Initials>I</Initials>
        </PersonNames>
        <Address>
          <Affiliation>Medizinische Hochschule Brandenburg, Institute for Health Professions Education Research, Neuruppin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Peters</Lastname>
          <LastnameHeading>Peters</LastnameHeading>
          <Firstname>Harm</Firstname>
          <Initials>H</Initials>
        </PersonNames>
        <Address>
          <Affiliation>Charit&#233; &#8211; Universit&#228;tsmedizin Berlin, Dieter Scheffner Center for Medical Education and Educational Research, Berlin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Kaminski</Lastname>
          <LastnameHeading>Kaminski</LastnameHeading>
          <Firstname>Julius J.</Firstname>
          <Initials>JJ</Initials>
        </PersonNames>
        <Address>
          <Affiliation>Charit&#233; &#8211; Universit&#228;tsmedizin Berlin, Dieter Scheffner Center for Medical Education and Educational Research, Berlin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
    </CreatorList>
    <PublisherList>
      <Publisher>
        <Corporation>
          <Corporatename>German Medical Science GMS Publishing House</Corporatename>
        </Corporation>
        <Address>D&#252;sseldorf</Address>
      </Publisher>
    </PublisherList>
    <SubjectGroup>
      <SubjectheadingDDB>610</SubjectheadingDDB>
    </SubjectGroup>
    <DatePublishedList>
      <DatePublished>20260904</DatePublished>
    </DatePublishedList>
    <Language>engl</Language>
    <License license-type="open-access" xlink:href="http://creativecommons.org/licenses/by/4.0/">
      <AltText language="en">This is an Open Access article distributed under the terms of the Creative Commons Attribution 4.0 License.</AltText>
      <AltText language="de">Dieser Artikel ist ein Open-Access-Artikel und steht unter den Lizenzbedingungen der Creative Commons Attribution 4.0 License (Namensnennung).</AltText>
    </License>
    <SourceGroup>
      <Meeting>
        <MeetingId>M0655</MeetingId>
        <MeetingSequence>105</MeetingSequence>
        <MeetingName>Jahrestagung der Gesellschaft f&#252;r Medizinische Ausbildung (GMA)</MeetingName>
        <MeetingTitle></MeetingTitle>
        <MeetingSession>P-01 Poster: KI-gest&#252;tztes Lehren und Lernen I</MeetingSession>
        <MeetingCity>Dresden</MeetingCity>
        <MeetingDate>
          <DateFrom>20260917</DateFrom>
          <DateTo>20260919</DateTo>
        </MeetingDate>
      </Meeting>
    </SourceGroup>
    <ArticleNo>P-0105</ArticleNo>
  </MetaData>
  <OrigData>
    <TextBlock name="Text" linked="yes">
      <MainHeadline>Text</MainHeadline><Pgraph><Mark1>Background: </Mark1>Clinical summary statements convert unstructured patient information into a concise format that supports initial diagnostic reasoning. In medical education, they are used to teach students how to organize, reduce, interpret, and synthesize clinical data <TextLink reference="1"></TextLink>. Assessing these summaries to provide effective feedback remains challenging. Previous attempts to automate this process relied on hard-coded techniques <TextLink reference="2"></TextLink>, but these approaches require substantial development effort. Large Language Models (LLMs) have shown strong performance in medical education and represent a promising alternative for automating the assessment of clinical summary statements.</Pgraph><Pgraph><Mark1>Methods: </Mark1>To assess the applicability of LLMs in scoring clinical summary statements, we conducted an experimental comparative study between four available LLMs (three large commercial models and a smaller local model). 122 summary statements from a dataset provided by Hege and colleagues <TextLink reference="2"></TextLink> were scored in five distinct runs. We assessed test-retest reliability using Fleiss&#8217; &#954; and Intraclass Correlation Coefficient (ICC) and concordance with human experts reported as inter-rater agreement using Gwet&#8217;s AC1&#47;AC2 and exact match accuracy. A rubric first proposed by Smith and colleagues <TextLink reference="3"></TextLink> and extended by Hege and colleagues was used as the basis for evaluation after annotating the rubric components for LLM use.</Pgraph><Pgraph><Mark1>Results: </Mark1>Four different prompting configurations were examined: zero-shot, zero-shot with chain-of-thought (CoT), few-shot, and few-shot CoT. All models demonstrated high test-retest reliability. GPT-4o and mistral large exhibited very high consistency while mistral 7B (the smaller local model) showed high ICC values but only fair to moderate agreement. Few-shot CoT prompting yielded the highest concordance with human expert raters, as reported by inter-rater agreement and exact match accuracy peaking for mistral large. Some rare exceptions and occasional extreme deviations were found in the small local mistral 7B model. Binary rubric components (person, factual accuracy) showed consistently higher agreement than ternary ones (transformation, global) (see figure 1 <ImgLink imgNo="1" imgType="figure" />).</Pgraph><Pgraph><Mark1>Discussion: </Mark1>LLMs can reliably rate clinical summary statements under appropriate prompting and with verbose and clear context data, such as annotated rubrics, enabling scalable automated feedback. Binary criteria showed higher agreement than ternary ones, indicating that LLMs excel at surface-level correctness and clearly defined features, while nuanced aspects of clinical reasoning remain a future opportunity. Human involvement is still recommended for oversight.</Pgraph><Pgraph><Mark1>Take-home message: </Mark1>LLMs can provide reliable, scalable feedback on clinical summaries, especially for clearly defined rubrics and criteria.</Pgraph></TextBlock>
    <References linked="yes">
      <Reference refNo="1">
        <RefAuthor>Feblowitz JC</RefAuthor>
        <RefAuthor>Wright A</RefAuthor>
        <RefAuthor>Singh H</RefAuthor>
        <RefAuthor>Samal L</RefAuthor>
        <RefAuthor>Sittig DF</RefAuthor>
        <RefTitle>Summarization of clinical information: A conceptual model</RefTitle>
        <RefYear>2011</RefYear>
        <RefJournal>J Biomed Inform</RefJournal>
        <RefPage>688-699</RefPage>
        <RefTotal>Feblowitz JC, Wright A, Singh H, Samal L, Sittig DF. Summarization of clinical information: A conceptual model. J Biomed Inform. 2011;44(4):688-699. DOI: 10.1016&#47;j.jbi.2011.03.008</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1016&#47;j.jbi.2011.03.008</RefLink>
      </Reference>
      <Reference refNo="2">
        <RefAuthor>Hege I</RefAuthor>
        <RefAuthor>Kiesewetter I</RefAuthor>
        <RefAuthor>Adler M</RefAuthor>
        <RefTitle>Automatic analysis of summary statements in virtual patients - a pilot study evaluating a machine learning approach</RefTitle>
        <RefYear>2020</RefYear>
        <RefJournal>BMC Med Educ</RefJournal>
        <RefPage>366</RefPage>
        <RefTotal>Hege I, Kiesewetter I, Adler M. Automatic analysis of summary statements in virtual patients - a pilot study evaluating a machine learning approach. BMC Med Educ. 2020;20(1):366. DOI: 10.1186&#47;s12909-020-02297-w</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1186&#47;s12909-020-02297-w</RefLink>
      </Reference>
      <Reference refNo="3">
        <RefAuthor>Smith S</RefAuthor>
        <RefAuthor>Kogan JR</RefAuthor>
        <RefAuthor>Berman NB</RefAuthor>
        <RefAuthor>Dell MS</RefAuthor>
        <RefAuthor>Brock DM</RefAuthor>
        <RefAuthor>Robins LS</RefAuthor>
        <RefTitle>The Development and Preliminary Validation of a Rubric to Assess Medical Students&#8217; Written Summary Statements in Virtual Patient Cases</RefTitle>
        <RefYear>2016</RefYear>
        <RefJournal>Acad Med</RefJournal>
        <RefPage>94-100</RefPage>
        <RefTotal>Smith S, Kogan JR, Berman NB, Dell MS, Brock DM, Robins LS. The Development and Preliminary Validation of a Rubric to Assess Medical Students&#8217; Written Summary Statements in Virtual Patient Cases. Acad Med. 2016;91(1):94-100. DOI: 10.1097&#47;ACM.0000000000000800</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1097&#47;ACM.0000000000000800</RefLink>
      </Reference>
    </References>
    <Media>
      <Tables>
        <NoOfTables>0</NoOfTables>
      </Tables>
      <Figures>
        <Figure width="566" height="675" format="png">
          <MediaNo>1</MediaNo>
          <MediaID>1</MediaID>
          <Caption><Pgraph><Mark1>Figure 1: Agreement values according to Gwet&#8217;s AC1&#47;AC2 by rubric component, prompting configuration, and model</Mark1><LineBreak></LineBreak>Binary components (person, factual accuracy) generally reached higher reliability than abstract ternary components (semantic qualifiers, transformation, narrowing, global). Agreement values are interpreted using the Landis and Koch (1977) scale, which ranges from slight to almost perfect agreement.</Pgraph></Caption>
        </Figure>
        <NoOfPictures>1</NoOfPictures>
      </Figures>
      <InlineFigures>
        <NoOfPictures>0</NoOfPictures>
      </InlineFigures>
      <Attachments>
        <NoOfAttachments>0</NoOfAttachments>
      </Attachments>
    </Media>
  </OrigData>
</GmsArticle>