<?xml version="1.0" encoding="iso-8859-1" standalone="no"?>
<!DOCTYPE GmsArticle SYSTEM "http://www.egms.de/dtd/2.0.34/GmsArticle.dtd">
<GmsArticle xmlns:xlink="http://www.w3.org/1999/xlink">
  <MetaData>
    <Identifier>mibe000311</Identifier>
    <IdentifierDoi>10.3205/mibe000311</IdentifierDoi>
    <IdentifierUrn>urn:nbn:de:0183-mibe0003116</IdentifierUrn>
    <ArticleType>Research Article</ArticleType>
    <TitleGroup>
      <Title language="en">Automatic extraction of clinically relevant parameters for tumor board decision support</Title>
      <TitleTranslated language="de">Automatische Extraktion klinisch relevanter Parameter zur Unterst&#252;tzung der Entscheidungsfindung im Tumorboard</TitleTranslated>
    </TitleGroup>
    <CreatorList>
      <Creator>
        <PersonNames>
          <Lastname>Y&#252;z&#252;nc&#252;oglu</Lastname>
          <LastnameHeading>Y&#252;z&#252;nc&#252;oglu</LastnameHeading>
          <Firstname>Imge</Firstname>
          <Initials>I</Initials>
        </PersonNames>
        <Address>Deutsches Forschungszentrum f&#252;r K&#252;nstliche Intelligenz (DFKI), Salzufer 15&#47;16, 10587 Berlin, Germany<Affiliation>German Research Centre for Artificial Intelligence (DFKI) Berlin, Germany</Affiliation></Address>
        <Email>imge.yuezuencueoglu&#64;dfki.de</Email>
        <Creatorrole corresponding="yes" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Chen</Lastname>
          <LastnameHeading>Chen</LastnameHeading>
          <Firstname>Yuxuan</Firstname>
          <Initials>Y</Initials>
        </PersonNames>
        <Address>
          <Affiliation>German Research Centre for Artificial Intelligence (DFKI) Berlin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>L&#252;ser</Lastname>
          <LastnameHeading>L&#252;ser</LastnameHeading>
          <Firstname>A. Altar</Firstname>
          <Initials>AA</Initials>
        </PersonNames>
        <Address>
          <Affiliation>German Research Centre for Artificial Intelligence (DFKI) Berlin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Ramasetti</Lastname>
          <LastnameHeading>Ramasetti</LastnameHeading>
          <Firstname>Nikitha Shruthi</Firstname>
          <Initials>NS</Initials>
        </PersonNames>
        <Address>
          <Affiliation>Charit&#233; &#8211; Universit&#228;tsmedizin Berlin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Oehring</Lastname>
          <LastnameHeading>Oehring</LastnameHeading>
          <Firstname>Robert</Firstname>
          <Initials>R</Initials>
        </PersonNames>
        <Address>
          <Affiliation>Charit&#233; &#8211; Universit&#228;tsmedizin Berlin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Krezien</Lastname>
          <LastnameHeading>Krezien</LastnameHeading>
          <Firstname>Felix</Firstname>
          <Initials>F</Initials>
        </PersonNames>
        <Address>
          <Affiliation>Charit&#233; &#8211; Universit&#228;tsmedizin Berlin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Thomas</Lastname>
          <LastnameHeading>Thomas</LastnameHeading>
          <Firstname>Philippe</Firstname>
          <Initials>P</Initials>
        </PersonNames>
        <Address>
          <Affiliation>German Research Centre for Artificial Intelligence (DFKI) Berlin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>M&#246;ller</Lastname>
          <LastnameHeading>M&#246;ller</LastnameHeading>
          <Firstname>Sebastian</Firstname>
          <Initials>S</Initials>
        </PersonNames>
        <Address>
          <Affiliation>German Research Centre for Artificial Intelligence (DFKI) Berlin, Germany</Affiliation>
          <Affiliation>Technische Universit&#228;t Berlin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Roller</Lastname>
          <LastnameHeading>Roller</LastnameHeading>
          <Firstname>Roland</Firstname>
          <Initials>R</Initials>
        </PersonNames>
        <Address>
          <Affiliation>German Research Centre for Artificial Intelligence (DFKI) Berlin, Germany</Affiliation>
        </Address>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
    </CreatorList>
    <PublisherList>
      <Publisher>
        <Corporation>
          <Corporatename>German Medical Science GMS Publishing House</Corporatename>
        </Corporation>
        <Address>D&#252;sseldorf</Address>
      </Publisher>
    </PublisherList>
    <SubjectGroup>
      <SubjectheadingDDB>610</SubjectheadingDDB>
      <Keyword language="en">natural language processing</Keyword>
      <Keyword language="en">information extraction</Keyword>
      <Keyword language="en">hepatocellular carcinoma</Keyword>
      <Keyword language="en">cholangiocarcinoma</Keyword>
      <Keyword language="en">colorectal liver metastases</Keyword>
      <Keyword language="en">sequence evaluation</Keyword>
      <Keyword language="en">tumor board</Keyword>
      <Keyword language="en">clinical decision support</Keyword>
      <Keyword language="de">nat&#252;rliche Sprachverarbeitung</Keyword>
      <Keyword language="de">Informationsextraktion</Keyword>
      <Keyword language="de">hepatozellul&#228;res Karzinom</Keyword>
      <Keyword language="de">Cholangiokarzinom</Keyword>
      <Keyword language="de">kolorektale Lebermetastasen</Keyword>
      <Keyword language="de">Sequenzauswertung</Keyword>
      <Keyword language="de">Tumorboard</Keyword>
      <Keyword language="de">Unterst&#252;tzung bei der klinischen Entscheidungsfindung</Keyword>
      <SectionHeading language="en">ISCB GMDS 2026</SectionHeading>
    </SubjectGroup>
    <DatePublishedList>
      <DatePublished>20260922</DatePublished>
    </DatePublishedList>
    <Language>engl</Language>
    <License license-type="open-access" xlink:href="http://creativecommons.org/licenses/by/4.0/">
      <AltText language="en">This is an Open Access article distributed under the terms of the Creative Commons Attribution 4.0 License.</AltText>
      <AltText language="de">Dieser Artikel ist ein Open-Access-Artikel und steht unter den Lizenzbedingungen der Creative Commons Attribution 4.0 License (Namensnennung).</AltText>
    </License>
    <SourceGroup>
      <Journal>
        <ISSN>1860-9171</ISSN>
        <Volume>22</Volume>
        <JournalTitle>GMS Medizinische Informatik, Biometrie und Epidemiologie</JournalTitle>
        <JournalTitleAbbr>GMS Med Inform Biom Epidemiol</JournalTitleAbbr>
      </Journal>
    </SourceGroup>
    <ArticleNo>13</ArticleNo>
    <Fundings>
      <Funding fundId="01VSF21047">Gemeinsamer Bundesausschuss</Funding>
      <Funding fundId="16KIS2048">Bundesministerium f&#252;r Forschung, Technologie und Raumfahrt (BMFTR)</Funding>
    </Fundings>
  </MetaData>
  <OrigData>
    <Abstract language="de" linked="yes"><Pgraph><Mark1>Einleitung:</Mark1> Die klinische Entscheidungsfindung st&#252;tzt sich auf Informationen, die in unstrukturierten Patientenberichten dokumentiert sind, die oft nicht standardisiert und nur schwer effizient zu verarbeiten sind. Diese Herausforderung ist besonders kritisch bei multidisziplin&#228;ren Tumorboard-Sitzungen, bei denen Entscheidungen unter Zeitdruck und auf Grundlage unvollst&#228;ndiger Informationen getroffen werden m&#252;ssen. Die automatisierte Informationsextraktion (IE), wie sie in <Mark1>ADBoard</Mark1> vorgestellt wird, unterst&#252;tzt die Strukturierung relevanter klinischer Daten f&#252;r die Entscheidungsfindung.</Pgraph><Pgraph><Mark1>Methoden:</Mark1> Wir haben eine IE-Pipeline f&#252;r Lebertumorf&#228;lle entwickelt und evaluiert, wobei der Schwerpunkt auf dem hepatozellul&#228;ren Karzinom, dem Cholangiokarzinom und kolorektalen Lebermetastasen lag. Ein Datensatz mit 743 klinischen Dokumenten von 134 Patienten wurde manuell annotiert, um Ground-Truth-Labels f&#252;r klinisch relevante Parameter zu erstellen. Die Aufgabe wurde als Named-Entity-Recognition (NER) formuliert. Wir verglichen regelbasierte Methoden (regul&#228;re Ausdr&#252;cke), Transformer-basierte Modelle und ein exploratives gro&#223;es Sprachmodell (LLM) unter realistischen klinischen Rahmenbedingungen.</Pgraph><Pgraph><Mark1>Ergebnisse:</Mark1> Transformer-basierte Modelle erzielten die beste Gesamtleistung, insbesondere bei h&#228;ufig vorkommenden Parametern. Regul&#228;re Ausdr&#252;cke schnitten bei klar definierten Mustern gut ab und erwiesen sich in Umgang mit begrenzten Ressourcen als robust. Das LLM zeigte eine uneinheitliche Leistung und konnte die anderen Methoden nicht &#252;bertreffen. Bei allen Ans&#228;tzen wurde die Leistung zudem durch Datenungleichgewichte und Schwankungen bei den Annotationen beeintr&#228;chtigt.</Pgraph><Pgraph><Mark1>Diskussion:</Mark1> Unsere Ergebnisse deuten darauf hin, dass unter den untersuchten realistischen, lokalen Einsatzbedingungen die explorative LLaMA-3.1-8B-Konfiguration bei der spezialisierten klinischen IE keine bessere Leistung erzielte als aufgabenspezifische Transformer- oder regelbasierte Ans&#228;tze. Stattdessen k&#246;nnte ein hybrider Ansatz, der Transformer-basierte und regelbasierte Methoden kombiniert, am effektivsten sein. Dar&#252;ber hinaus hat die Bewertungsmethodik einen erheblichen Einfluss auf die Interpretation der Leistung, da <Mark2>strenge</Mark2>, auf Sequenzen basierende Metriken klinisch relevante Ergebnisse unterbewerten k&#246;nnen. Diese Ergebnisse unterstreichen, dass die Verbesserung der Annotationsqualit&#228;t und des Bewertungsdesigns f&#252;r die klinische IE ebenso wichtig ist wie Fortschritte in der Modellarchitektur.</Pgraph></Abstract>
    <Abstract language="en" linked="yes"><Pgraph><Mark1>Introduction:</Mark1> Clinical decision-making relies on information documented in unstructured reports, which are often non-standardized and difficult to process efficiently. This challenge is particularly critical in multidisciplinary tumor board meetings, where decisions must be made under time constraints and based on incomplete information. Automated information extraction (IE) as presented in <Mark1>ADBoard</Mark1> supports structuring relevant clinical data for decision-making.</Pgraph><Pgraph><Mark1>Methods:</Mark1> We developed and evaluated an IE pipeline for liver tumor cases, focusing on hepatocellular carcinoma, cholangiocarcinoma, and colorectal liver metastases. A dataset of 743 clinical documents from 134 patients was manually annotated to create ground truth labels for clinically relevant parameters. The task was formulated as named entity recognition (NER). We compared rule-based methods (regular expressions), transformer-based models, and an exploratory large language model (LLM) under realistic clinical constraints.</Pgraph><Pgraph><Mark1>Results:</Mark1> Transformer-based models achieved the strongest overall performance, particularly for frequent parameters. Regular expressions performed competitively for well-defined patterns and proved robust in low-resource settings. The LLM showed inconsistent performance and did not outperform the other methods. Across all approaches, performance was also affected by data imbalance and annotation variability.</Pgraph><Pgraph><Mark1>Discussion:</Mark1> Our findings indicate that, under the evaluated realistic, local deployment constraints, the exploratory LLaMA 3.1 8B setup did not outperform task-specific transformer or rule-based approaches for specialized clinical IE. Instead, a hybrid approach combining transformer-based and rule-based methods might be most effective. Furthermore, evaluation methodology substantially influences performance interpretation, as <Mark2>strict</Mark2> span-based metrics can underestimate clinically relevant results. These results highlight that improving annotation quality and evaluation design is as important as advances in model architecture for clinical IE.</Pgraph></Abstract>
    <TextBlock name="1 Introduction" linked="yes">
      <MainHeadline>1 Introduction</MainHeadline><Pgraph>Within the medical domain, patient information, such as clinical condition, treatments, and medication, is often documented in non-standardized, unstructured reports. Although these reports contain essential information for clinical decision-making, extracting relevant data remains time-consuming and error-prone. This challenge is particularly critical in multidisciplinary tumor board meetings, where treatment decisions must be made under time constraints and often with incomplete information due to administrative or procedural reasons. To support physicians in this setting, we developed <Mark1>ADBoard</Mark1>, a clinical decision support system for liver tumor treatment recommendations <TextLink reference="1"></TextLink>. The system targets hepatocellular carcinoma, cholangiocarcinoma, and colorectal liver metastases. It extracts clinically relevant parameters from unstructured reports and transforms them into a structured and consolidated tumor board protocol, complemented by treatment recommendations. Some of these parameters are tumor-specific findings (e.g., <Mark2>tumor size</Mark2>, <Mark2>tumor count</Mark2>, and <Mark2>TNM staging</Mark2>) as well as clinically relevant factors such as <Mark2>ascites</Mark2>, and <Mark2>vascular invasion</Mark2>, which are routinely considered during tumor board decision-making. The extracted information is presented in form of a normalized, interpretable tumor board protocol to support decision-making during tumor board discussions <TextLink reference="2"></TextLink>.</Pgraph><Pgraph>In this work, we focus on the information extraction (IE) component, which aims to derive structured information from unstructured clinical text. As we extract predefined clinical parameters, we formulate the task as named entity recognition (NER), where entities of interest are identified and extracted <TextLink reference="3"></TextLink>, <TextLink reference="4"></TextLink>. We compare rule-based, transformer-based, and an exploratory large language model (LLM)-based approaches under realistic clinical constraints. Our contributions are as follows: </Pgraph><Pgraph><OrderedList><ListItem level="1" levelPosition="1" numString="1.">a comparative evaluation of rule-based, transformer-based, and LLM-based approaches for German clinical IE; </ListItem><ListItem level="1" levelPosition="2" numString="2.">an analysis of the role of annotation quality in clinical extraction tasks; and </ListItem><ListItem level="1" levelPosition="3" numString="3.">a discussion of practical limitations of current approaches in real-world settings.</ListItem></OrderedList></Pgraph></TextBlock>
    <TextBlock name="2 Tumor board data" linked="yes">
      <MainHeadline>2 Tumor board data</MainHeadline><SubHeadline2>Data description</SubHeadline2><Pgraph>Treatment decisions for liver tumor patients are based on a set of clinically relevant parameters, including tumor-independent factors (e.g., <Mark2>cirrhosis</Mark2>) and tumor-specific characteristics that vary across mentioned tumor types. Identifying these parameters is essential for guideline-compliant decision-making in tumor board settings. To extract such information from unstructured clinical reports, we iteratively defined a set of relevant parameters motivated by the guideline program <TextLink reference="5"></TextLink> together with clinical experts. The parameters include numerical values (e.g., <Mark2>tumor size</Mark2>, <Mark2>number of tumors</Mark2>), binary indicators (e.g., presence of <Mark2>cirrhosis</Mark2>), and categorical or free-text information (e.g., <Mark2>TNM staging</Mark2>). In total, 31 parameters, represented in Table 1 <ImgLink imgNo="1" imgType="table" />, were defined. Due to class imbalance, varying clinical relevance, and limited representation, we mainly focus on a subset of 14 key parameters (highlighted in Table 1 <ImgLink imgNo="1" imgType="table" />) that are consistently relevant across the three mentioned tumor types. These parameters were selected based on two criteria: </Pgraph><Pgraph><OrderedList><ListItem level="1" levelPosition="1" numString="1.">their relevance for treatment decisions according to clinical guidelines and tumor board workflows, and </ListItem><ListItem level="1" levelPosition="2" numString="2.">expert assessment by participating physicians regarding their practical importance and informational value in routine clinical decision-making.</ListItem></OrderedList></Pgraph><Pgraph><ImgPlaceholder imgNo="1" imgType="table"/></Pgraph><SubHeadline2>Data generation</SubHeadline2><Pgraph>We constructed a pseudonymized German clinical dataset, approved for scientific use by the relevant ethics committee (EA4&#47;169&#47;22). Development started with hepatocellular carcinoma due to higher data availability, starting with an initial corpus of 144 unstructured clinical documents, including discharge letters, radiology reports, pathology reports, and surgical reports. For experiments, we focused on radiology and pathology reports, as these are available in the targeted real-time clinical decision support setting. To enable training and evaluation of extraction methods, the documents were manually annotated to create a reference dataset, i.e. clinically relevant parameters were explicitly marked in the text, defining the expected correct information (&#8220;ground truth&#8221;). Annotation was performed by a single annotator using INCEpTION <TextLink reference="6"></TextLink>. Hence, annotation omissions, boundary inconsistencies, and differences in annotation interpretation may have influenced both model training and evaluation. Due to class imbalance, with some parameters occurring fewer than five times, additional radiology reports were incorporated iteratively as they became available and were annotated using the same annotation scheme and guidelines. The final dataset comprises 743 documents from 134 patients (581 radiology, 162 pathology reports). As some parameters were introduced at later stages, previously annotated documents were not retrospectively re-annotated for these parameters. Consequently, not all documents contain annotations for all parameters, which is considered in the error analysis.</Pgraph></TextBlock>
    <TextBlock name="3 Methodology" linked="yes">
      <MainHeadline>3 Methodology</MainHeadline><Pgraph>This section provides an overview of the experimental setup and the evaluated IE approaches. After generating the dataset and performing some preprocessing, we compared three methodological paradigms for clinical IE: regular expression (RegEx) patterns, transformer-based sequence labelling models, and an exploratory prompt-based large language model (LLM) baseline. All approaches are trained and evaluated using the same data splits and evaluation framework, enabling a consistent comparison of symbolic, statistical, and generative methods for extracting clinical IE.</Pgraph><SubHeadline>3.1 Preprocessing</SubHeadline><Pgraph>To enable consistent processing and comparison across different extraction methods, the dataset was preprocessed prior to experimentation. Sentence segmentation by INCEpTION was refined using RegEx postprocessing to account for frequent clinical abbreviations (e.g., <Mark2>Pat</Mark2>., <Mark2>klin</Mark2>.), which can otherwise lead to incorrect sentence boundaries. All annotations were converted into the BIO tagging scheme to support token-level sequence labeling and ensure comparability across approaches <TextLink reference="4"></TextLink>. Finally, the dataset was split at document level into training (47<TextGroup><PlainText>6 d</PlainText></TextGroup>ocuments), development (119 documents), and test (148 documents) sets. Rather than using a purely proportional split, we aimed to ensure sufficient representation of relevant parameters in each subset. A patient-level split was not applied because, given the limited cohort size and the low prevalence of several parameters, this would have resulted in some extraction targets being insufficiently represented or absent from individual subsets, limiting their suitability for model development and evaluation. Consequently, 59 of 134 patients (44.0&#37;) occur in more than one of the training, development, and test subsets. This patient overlap may lead to optimistic performance estimates and limits conclusions regarding generalization to previously unseen patients. Despite efforts to balance the parameters across the subsets, the data remains naturally imbalanced, with some parameters (e.g. portal hypertension) being underrepresented (see Table 1 <ImgLink imgNo="1" imgType="table" />).</Pgraph><SubHeadline>3.2 Symbolic, statistical, and generative IE approaches</SubHeadline><SubHeadline2>Regular expressions</SubHeadline2><Pgraph>RegEx are a well-established method for IE <TextLink reference="3"></TextLink>. Patterns were iteratively developed exclusively on the training set using domain knowledge from clinical experts and inspection of annotated data. To improve precision, document structure was incorporated by restricting extraction to relevant sections (e.g., <Mark2>diagnosis, microscopy, macroscopy</Mark2>). In total, 38 patterns were defined. For negation detection, used to determine the existence or non-existence of binary parameters such as <Mark2>ascites</Mark2> and <Mark2>vascular invasion</Mark2>, we implemented <Mark1>pynegex</Mark1> (<Hyperlink href="https:&#47;&#47;pypi.org&#47;project&#47;pynegex&#47;">https:&#47;&#47;pypi.org&#47;project&#47;pynegex&#47;</Hyperlink>). Documents were processed at the sentence level, matching each sentence against the defined patterns.</Pgraph><SubHeadline2>Transformer models</SubHeadline2><Pgraph>Transformer-based models, particularly BERT variants, are widely used for NER tasks and are often state-of-the-art methods <TextLink reference="7"></TextLink>, <TextLink reference="8"></TextLink>, <TextLink reference="9"></TextLink>. We evaluate three models: </Pgraph><Pgraph><OrderedList><ListItem level="1" levelPosition="1" numString="1.">google-bert&#47;bert-base-german-cased <TextLink reference="10"></TextLink> as a German general-domain baseline, </ListItem><ListItem level="1" levelPosition="2" numString="2.">GerMedBERT&#47;medbert-512 <TextLink reference="11"></TextLink> as a German domain-specific medical BERT variant, and </ListItem><ListItem level="1" levelPosition="3" numString="3.">FacebookAI&#47;xlm-roberta-base <TextLink reference="12"></TextLink> as a multilingual transformer model supporting German, </ListItem></OrderedList></Pgraph><Pgraph>hereafter referred to as BERT, MedBERT, and XLM-RoBERTa, respectively. All models were trained on the training set and tuned on a development set using AdamW optimization, cross-entropy loss, and early stopping based on F1 score (patience of six epochs). Input sequences were truncated or padded to the maximum model length. Hyperparameter tuning considered learning rates of 3e-5 and 5e-5, batch sizes of 16 and 32, and maximum sequence lengths of 256 and 512 tokens. To assess the impact of input granularity, experiments were conducted at both sentence and document level. Sentence-level processing treats each sentence independently, whereas document-level processing uses entire reports to capture broader contextual dependencies. The best-performing configuration for each transformer model was selected based on development-set performance and subsequently evaluated on the test set.</Pgraph><SubHeadline2>Exploratory LLM baseline</SubHeadline2><Pgraph>Model selection was constrained by two requirements: </Pgraph><Pgraph><OrderedList><ListItem level="1" levelPosition="1" numString="1.">All processing had to be performed locally due to sensitive patient data, and </ListItem><ListItem level="1" levelPosition="2" numString="2.">models had to be deployable in real-world clinical settings, limiting the use of large models. </ListItem></OrderedList></Pgraph><Pgraph>We therefore evaluated LLaMA 3.1 8B (LLaMA) <TextLink reference="13"></TextLink>, a lightweight open-source model suitable for local deployment. Documents were provided as full input to enable the model to leverage document-level context. Prompts were manually designed as German <Mark1>zero-shot prompts</Mark1> for each extraction parameter. They instruct the model to act as an experienced German physician reviewing clinical reports in a tumor board context. For each parameter, a separate inference call was performed on the full document, and the model was asked to extract the relevant information and return only JSON output. Inference was performed locally with a temperature of 0 and a maximum generation length of 512 tokens. Other decoding parameters were left at the backend defaults. For example, for tumor size extraction, the model was instructed to return a list under the key TSIZE, with each entry containing the extracted size including its unit and the original sentence from which it was extracted. If no tumor size was found, the model was instructed to return an empty list. Generated outputs were subsequently parsed, validated for the expected parameter format, and mapped back to BIO spans for evaluation.</Pgraph><SubHeadline>3.3 Evaluation metrics</SubHeadline><Pgraph>Evaluation of NER in clinical text is challenging due to inconsistencies, particularly in span boundaries (e.g., missing units such as mm or cm). To address this, we employ three evaluation strategies: <Mark2>strict</Mark2>, <Mark2>lenient</Mark2>, and <Mark2>manual</Mark2>, reporting performance using the F1 score, as it provides a balanced measure of precision and recall, both important in the clinical domain. The <Mark2>strict</Mark2> evaluation requires exact matches between prediction and ground truth and is highly sensitive to boundary deviations. In contrast, <Mark2>lenient</Mark2> evaluation measures character-level overlap using the <Mark2>Jaccard Index</Mark2> <TextLink reference="14"></TextLink>. For the 14 key parameters, we additionally conducted a <Mark2>manual</Mark2> evaluation based on clinically validated reference values provided by the participating physicians. Predictions were primarily reviewed by the computer scientists and considered correct if the relevant clinical information matched these reference values, irrespective of exact span boundaries. Ambiguous cases were discussed with the clinical experts, whose assessment was considered decisive. These complementary evaluation strategies account for annotation variability and provide a more realistic assessment of IE performance in clinical settings.</Pgraph></TextBlock>
    <TextBlock name="4 Results discussion and error analysis" linked="yes">
      <MainHeadline>4 Results discussion and error analysis</MainHeadline><SubHeadline>4.1 Transformer model comparison</SubHeadline><Pgraph>Figure 1 <ImgLink imgNo="1" imgType="figure" /> summarizes the test set performance of the evaluated transformer models. MedBERT achieved the highest aggregate F1 scores (51&#37; at sentence level and 47&#37; at document level). All models achieved comparable performance, with the largest difference between MedBERT and XLM-RoBERTa at sentence level being 5 percentage points only. Across all transformer variants, sentence-level IE slightly outperformed document-level IE, suggesting that shorter and more focused contexts may facilitate parameter extraction compared to longer and potentially noisier reports. However, the results should be interpreted considering the dataset characteristics: Several parameters were underrepresented, making performance estimates sensitive to individual prediction errors and class imbalance.</Pgraph><SubHeadline>4.2 Per parameter evaluation</SubHeadline><Pgraph>The detailed per-parameter results are shown in Table 1 <ImgLink imgNo="1" imgType="table" /> for XLM-RoBERTa, RegEx, and LLaMA. While Figure 1 <ImgLink imgNo="1" imgType="figure" /> shows that MedBERT achieves the strongest aggregate transformer performance, the differences between the transformer models are relatively small. XLM-RoBERTa is used as the representative transformer model in Table 1 <ImgLink imgNo="1" imgType="table" /> because detailed manually reviewed parameter-level results were available for this model. Thus, Figure 1 <ImgLink imgNo="1" imgType="figure" /> provides the overall transformer comparison, whereas <TextGroup><PlainText>Table 1 </PlainText></TextGroup><ImgLink imgNo="1" imgType="table" /> enables a detailed parameter-level comparison with RegEx and LLaMA. The LLM evaluation is restricted to the subset of 14 focused parameters. </Pgraph><SubHeadline2>Comparing evaluation methods</SubHeadline2><Pgraph>We observe substantial discrepancies between <Mark2>strict</Mark2>, <Mark2>lenient</Mark2>, and <Mark2>manual</Mark2> evaluation, indicating that the choice of metric strongly affects performance interpretation. For example, XLM-RoBERTa, RegEx, and LLaMA achieve F1 scores of 37&#37;, 21&#37;, and 34&#37; under <Mark2>strict</Mark2> evaluation for IE of <Mark2>ascites</Mark2>, compared to 71&#37;, 56&#37;, and 67&#37; under <Mark2>lenient</Mark2> and 83&#37;, 75&#37;, and 86&#37; under <Mark2>manual</Mark2> evaluation. Across the 14 parameters, differences of at least ten percentage points between <Mark2>strict</Mark2> and <Mark2>manual</Mark2> evaluation are observed for most parameters. <Mark2>Strict</Mark2> span-based evaluation tends to underestimate clinically relevant performance, while <Mark2>manual</Mark2> evaluation better reflects practical utility. <Mark2>Lenient</Mark2> evaluation provides an intermediate perspective by partially rewarding overlapping spans as they frequently lie between <Mark2>strict</Mark2> and <Mark2>manual</Mark2> results and do not consistently approximate either. These results highlight that evaluation protocols are a critical factor in assessing clinical IE systems. In settings with free-text data and non-standardized annotations, <Mark2>strict</Mark2> span-based metrics may not adequately reflect clinically meaningful performance. This points to an open challenge in clinical NLP: designing evaluation schemes that capture relevant information beyond exact span boundaries.</Pgraph><SubHeadline2>Comparing F1 scores</SubHeadline2><Pgraph>Several parameters are underrepresented in the dataset, making evaluation sensitive to individual prediction errors. Based on manual evaluation, LLaMA performs slightly better than XLM-RoBERTa and RegEx for only two parameters <Mark2>(ascites</Mark2> and <Mark2>cirrhosis</Mark2>), but the differences are marginal (3 and 1 percentage points) and do not indicate a meaningful advantage. Overall, XLM-RoBERTa achieves the best performance across most parameters, outperforming RegEx on nine out of 14 key parameters. However, results vary considerably depending on data availability. For example, RegEx outperforms XLM-RoBERTa for <Mark2>portal hypertension</Mark2>, but this parameter occurs only twice, limiting the reliability of this observation. This highlights the strong dependence of transformer-based models on data quantity and quality. Frequently occurring parameters are learned effectively, whereas rare parameters remain challenging. In our exploratory LLM setup, zero-shot prompt-based IE did not outperform the transformer or rule-based approaches. Additionally, results were sensitive to prompt formulation, limiting reproducibility, and the model&#8217;s opacity made error analysis and controlled improvements difficult. These results should be interpreted as evidence of limitations of the local deployment scenario, rather than as a general conclusion about the reliability of LLMs for clinical IE.</Pgraph><SubHeadline>4.3 Error analysis</SubHeadline><Pgraph>To better understand the limitations of the evaluated approaches, we conducted a comprehensive error analysis during manual evaluation. The error categories identified were found to be consistent across transformer-based approaches and are therefore discussed at the methodological level rather than for a single transformer variant. The analysis has shown that model limitations, parameter-specific errors, and annotation errors contribute to incorrect predictions.</Pgraph><SubHeadline2>Model limitations</SubHeadline2><Pgraph>The approaches exhibit different limitations. RegEx performs well for structured patterns but lacks contextual understanding, resulting in false positives for ambiguous terms (e.g., <Mark2>Fl&#252;ssigkeit</Mark2> (fluids) but not related to <Mark2>ascites)</Mark2>. Transformer models, in contrast, struggle with rare parameters and complex clinical language, particularly in cases requiring contextual interpretation such as negation or implicit references.</Pgraph><SubHeadline2>Representative parameter-level errors</SubHeadline2><Pgraph>Certain parameters are particularly prone to errors. <Mark2>Tumor size</Mark2> is often over-predicted, as models incorrectly label related measurements (e.g., <Mark2>organ size</Mark2> or <Mark2>resection size</Mark2>) as <Mark2>tumor size</Mark2>. Binary parameters like <Mark2>ascites</Mark2> and <Mark2>cirrhosis</Mark2> are frequently affected by negation errors, leading to false positives despite explicit negation. <Mark2>Distant metastasis</Mark2> remains challenging due to ambiguous, context-dependent expressions requiring additional contextual information beyond annotated spans. </Pgraph><SubHeadline2>Annotation-related errors</SubHeadline2><Pgraph>Annotation quality may have influenced both model training and evaluation. During manual inspection of prediction errors, we observed cases in which model outputs appeared clinically plausible despite the absence of corresponding annotations. For example, the phrase <Mark2>Keine freie Fl&#252;ssigkeit im kleinen Becken</Mark2> (No free fluid in the pelvis) was predicted to be relevant to <Mark2>ascites</Mark2>, although no corresponding annotation was present in the reference dataset. This particularly affected parameters introduced later in the annotation process, as earlier documents had not been retrospectively re-annotated. Consequently, missing annotations could not always be distinguished from true negative labels during automated training and evaluation. While such observations do not allow conclusions about the overall annotation quality, they suggest that annotation omissions may contribute to some apparent false positives. We further observed inconsistencies related to span boundaries and negation handling. In some cases, negated expressions were annotated without including the negation cue itself, which may lead to mismatches under <Mark2>strict</Mark2> span-level evaluation. Consequently, part of the observed error rate may reflect differences between model predictions and annotation conventions rather than purely incorrect IE. All in all, the observed errors likely reflect a combination of model limitations, annotation related factors, data imbalance, and the inherent complexity of clinical language.</Pgraph></TextBlock>
    <TextBlock name="5 Related work" linked="yes">
      <MainHeadline>5 Related work</MainHeadline><Pgraph>Clinical IE has traditionally been addressed using rule-based and machine learning based approaches. Earlier studies show that rule-based systems dominated the field <TextLink reference="15"></TextLink>, while more recent work highlights persistent challenges such as limited annotated data, restricted access to clinical corpora, and difficulties in model sharing <TextLink reference="16"></TextLink>. Even with domain-adapted models, improvements remain moderate as there are reports of only around 6&#37; improvement for medical decision support tasks <TextLink reference="8"></TextLink>. A fundamental limitation of many IE approaches is their focus on sentence-level extraction, neglecting relations across sentences and documents. However, clinical narratives often distribute relevant information across multiple sentences or reports, requiring contextual aggregation. Prior work has addressed this using methods such as reference resolution to link related mentions across text <TextLink reference="17"></TextLink>. This improves downstream extraction of tumor characteristics but remains challenging and achieves only moderate performance. Similarly, transformer-based approaches using question-answering formulations have been applied for IE <TextLink reference="18"></TextLink>. They demonstrate strong performance for explicit information (e.g., <Mark2>age</Mark2>, <Mark2>gender</Mark2>, <Mark2>histology</Mark2>), but limitations for more complex attributes such as <Mark2>tumor location</Mark2>. </Pgraph><Pgraph>Recent work with LLMs investigates their ability to capture contextual and temporal relations. While they perform well on benchmark datasets <TextLink reference="19"></TextLink>, their effectiveness in real-world clinical settings remains limited, particularly for long and fragmented patient records. Therefore, LLMs are being combined with Retrieval-Augmented Generation (RAG) and structured representations <TextLink reference="20"></TextLink>. For instance, CliCARE transforms longitudinal electronic health records into temporal knowledge graphs to model temporal dependencies and align them with clinical guidelines, significantly outperforming standard RAG approaches. However, RAG approaches remain limited in reliability with reports that 17&#37; of generated references are hallucinated, indicating that such systems are not yet sufficiently robust for clinical deployment <TextLink reference="21"></TextLink>. </Pgraph><Pgraph>All in all, prior work shows that clinical IE requires not only accurate NER but also modeling of contextual and temporal relations. Despite recent advances, these challenges remain partially solved, particularly under realistic constraints such as limited data and domain-specific settings. Our findings are consistent with this: in a German clinical IE setting under realistic constraints, the LLM approach does not outperform transformer or rule-based methods, highlighting the need for more robust approaches to clinical IE.</Pgraph></TextBlock>
    <TextBlock name="6 Limitations" linked="yes">
      <MainHeadline>6 Limitations</MainHeadline><Pgraph>This study has several limitations: </Pgraph><Pgraph><OrderedList><ListItem level="1" levelPosition="1" numString="1.">We had only one medical specialist as an annotator. Therefore, no inter-annotator agreement could be assessed. Annotation noise, missing labels, and boundary inconsistencies may affect both model training and evaluation and limit the reliability of automatic metrics. </ListItem><ListItem level="1" levelPosition="2" numString="2.">The dataset is relatively small and imbalanced, with some parameters occurring rarely, restricting the ability of data-driven models to generalize and making performance comparisons for rare parameters less reliable. Furthermore, the document-level splitting strategy resulted in 59 of 134 patients occurring in more than one subset, which may lead to optimistic performance estimates and limits conclusions regarding generalization to unseen patients. </ListItem><ListItem level="1" levelPosition="3" numString="3.">The study is conducted on German clinical text, where publicly available resources are limited due to data protection constraints. Although this increases practical relevance, it reduces comparability with prior work, which is predominantly based on English datasets. </ListItem><ListItem level="1" levelPosition="4" numString="4.">RegEx patterns and LLM prompts were manually developed and are therefore sensitive to design choices, limiting reproducibility and generalizability to other datasets and clinical settings. </ListItem><ListItem level="1" levelPosition="5" numString="5.">Due to resource constraints, detailed parameter-level evaluation was conducted only for a subset of models and parameters. </ListItem><ListItem level="1" levelPosition="6" numString="6.">The requirement for local deployability in a realistic setting constrained the evaluation to a single exemplary LLM-based approach. Therefore, the findings cannot be generalized to LLMs in general, particularly to larger models that could not be considered under these requirements. </ListItem><ListItem level="1" levelPosition="7" numString="7.">Finally, this work evaluates the IE component of ADBoard retrospectively and does not include prospective validation within actual tumor board meetings. Such workflow-level validation is outside the scope of this present NER-focused work and is being investigated separately.</ListItem></OrderedList></Pgraph></TextBlock>
    <TextBlock name="7 Conclusion" linked="yes">
      <MainHeadline>7 Conclusion</MainHeadline><Pgraph>In this work, we investigate IE for clinical decision support as part of the <Mark1>ADBoard</Mark1> pipeline. Using a real-world dataset, we compared rule-based, transformer-based, and an exploratory LLM approach under real-world clinical conditions. Our results show that transformer-based models achieve the strongest overall performance, while RegEx approaches can provide a resource-efficient complement for well-structured parameters. In contrast, the LLM-based approach does not outperform existing methods in this setting. We further demonstrate that evaluation methodology strongly influences performance interpretation, as strict span-based metrics can underestimate clinically relevant results. In practice, a hybrid approach combining transformer-based and rule-based methods could be most effective, leveraging the strengths of both paradigms depending on the parameter and data availability. Furthermore, our findings suggest that annotation quality and evaluation design are as important as advances in model architecture for clinical IE. Future work should focus on improving annotation, a better distribution of the parameters, exploring more advanced LLM-based approaches, and developing evaluation schemes that better reflect clinically meaningful extraction performance.</Pgraph></TextBlock>
    <TextBlock name="Notes" linked="yes">
      <MainHeadline>Notes</MainHeadline><SubHeadline>Author contributions</SubHeadline><Pgraph>RR, SM, PT, FK: Conception and design of the study; NSR, RO: Collecting, preparing and annotating data; IY, YC, AAL: Implementation and conduction of experiments; YC, IY, AAL, RO, FK: Evaluation; IY: Writing of manuscript; RR, AAL, YC, PT, SM, RO, FK, NSR: Review of manuscript. All authors have approved the manuscript as submitted and assume responsibility for the scientific integrity of the work. The authors declare that there is no conflict of interest.</Pgraph><SubHeadline>Acknowledgements</SubHeadline><Pgraph>This work was carried out as part of the <Mark1>ADBoard</Mark1> project (01VSF21047) funded by the Joint Federal Committee and supported by the Federal Ministry of Research, Technology and Space (BMFTR) through the project <TextGroup><Mark1>Veranda</Mark1></TextGroup> (16KIS2048).</Pgraph><SubHeadline>Competing interests</SubHeadline><Pgraph>The authors declare that they have no competing interests.</Pgraph></TextBlock>
    <References linked="yes">
      <Reference refNo="1">
        <RefAuthor>Ng SST</RefAuthor>
        <RefAuthor>Oehring R</RefAuthor>
        <RefAuthor>Ramasetti N</RefAuthor>
        <RefAuthor>Roller R</RefAuthor>
        <RefAuthor>Thomas P</RefAuthor>
        <RefAuthor>Chen Y</RefAuthor>
        <RefAuthor>Moosburner S</RefAuthor>
        <RefAuthor>Winter A</RefAuthor>
        <RefAuthor>Maurer MM</RefAuthor>
        <RefAuthor>Auer TA</RefAuthor>
        <RefAuthor>Kamali C</RefAuthor>
        <RefAuthor>Pratschke J</RefAuthor>
        <RefAuthor>Benzing C</RefAuthor>
        <RefAuthor>Krenzien F</RefAuthor>
        <RefTitle>Concordance of a decision algorithm and multidisciplinary team meetings for patients with liver cancer-a study protocol for a randomized controlled trial</RefTitle>
        <RefYear>2023</RefYear>
        <RefJournal>Trials</RefJournal>
        <RefPage>577</RefPage>
        <RefTotal>Ng SST, Oehring R, Ramasetti N, Roller R, Thomas P, Chen Y, Moosburner S, Winter A, Maurer MM, Auer TA, Kamali C, Pratschke J, Benzing C, Krenzien F. Concordance of a decision algorithm and multidisciplinary team meetings for patients with liver cancer-a study protocol for a randomized controlled trial. Trials. 2023 Sep 9;24(1):577. DOI: 10.1186&#47;s13063-023-07610-8</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1186&#47;s13063-023-07610-8</RefLink>
      </Reference>
      <Reference refNo="2">
        <RefAuthor>Y&#252;z&#252;nc&#252;oglu Y</RefAuthor>
        <RefAuthor>Chen Y</RefAuthor>
        <RefAuthor>L&#252;ser AA</RefAuthor>
        <RefAuthor>Roller R</RefAuthor>
        <RefAuthor>Thomas P</RefAuthor>
        <RefAuthor>M&#246;ller S</RefAuthor>
        <RefAuthor>Ramasetti NS</RefAuthor>
        <RefAuthor>Ng SST</RefAuthor>
        <RefAuthor>Oehring R</RefAuthor>
        <RefAuthor>Krenzien F</RefAuthor>
        <RefTitle></RefTitle>
        <RefYear></RefYear>
        <RefBookTitle>Entwicklung einer automatisierten Informationsextraktion und Entscheidungsunterst&#252;tzung f&#252;r die Tumorkonferenz hepatozellul&#228;rer Karzinome. KI in der Arzt-Patienten-Kommunikation</RefBookTitle>
        <RefPage></RefPage>
        <RefTotal>Y&#252;z&#252;nc&#252;oglu Y, Chen Y, L&#252;ser AA, Roller R, Thomas P, M&#246;ller S, Ramasetti NS, Ng SST, Oehring R, Krenzien F. Entwicklung einer automatisierten Informationsextraktion und Entscheidungsunterst&#252;tzung f&#252;r die Tumorkonferenz hepatozellul&#228;rer Karzinome. KI in der Arzt-Patienten-Kommunikation. Forthcoming.</RefTotal>
      </Reference>
      <Reference refNo="3">
        <RefAuthor>Jurafsky D</RefAuthor>
        <RefAuthor>Martin JH</RefAuthor>
        <RefTitle></RefTitle>
        <RefYear>2026</RefYear>
        <RefBookTitle>Speech and language processing: an introduction to natural language processing, computational linguistics, and speech recognition, with language models</RefBookTitle>
        <RefPage></RefPage>
        <RefTotal>Jurafsky D, Martin JH. Speech and language processing: an introduction to natural language processing, computational linguistics, and speech recognition, with language models. 3rd ed. draft. 2026 Jan 6. Available from: https:&#47;&#47;web.stanford.edu&#47;&#126;jurafsky&#47;slp3&#47;</RefTotal>
        <RefLink>https:&#47;&#47;web.stanford.edu&#47;&#126;jurafsky&#47;slp3&#47;</RefLink>
      </Reference>
      <Reference refNo="4">
        <RefAuthor>Alshammari N</RefAuthor>
        <RefAuthor>Alanazi S</RefAuthor>
        <RefTitle>The impact of using different annotation schemes on named entity recognition</RefTitle>
        <RefYear>2021</RefYear>
        <RefJournal>Egyptian Informatics Journal</RefJournal>
        <RefPage>295-302</RefPage>
        <RefTotal>Alshammari N, Alanazi S. The impact of using different annotation schemes on named entity recognition. Egyptian Informatics Journal. 2021;22(3):295-302. DOI: 10.1016&#47;j.eij.2020.10.004</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1016&#47;j.eij.2020.10.004</RefLink>
      </Reference>
      <Reference refNo="5">
        <RefAuthor>Leitlinienprogramm Onkologie</RefAuthor>
        <RefTitle></RefTitle>
        <RefYear>2025</RefYear>
        <RefBookTitle>3-Leitlinie Diagnostik und Therapie des hepatozellul&#228;ren Karzinoms und bili&#228;rer Karzinome. AWMF-Registernummer 032-053OL</RefBookTitle>
        <RefPage></RefPage>
        <RefTotal>Leitlinienprogramm Onkologie. S3-Leitlinie Diagnostik und Therapie des hepatozellul&#228;ren Karzinoms und bili&#228;rer Karzinome. AWMF-Registernummer 032-053OL. Version 5.2. Langfassung. 2025 Jun &#91;accessed 2026 Mar 31&#93;. Available from: https:&#47;&#47;www.leitlinienprogramm-onkologie.de&#47;leitlinien&#47;hcc-und-biliaere-karzinome</RefTotal>
        <RefLink>https:&#47;&#47;www.leitlinienprogramm-onkologie.de&#47;fileadmin&#47;user&#95;upload&#47;Downloads&#47;Leitlinien&#47;HCC&#47;Version&#95;5&#47;LL&#95;Hepatozellul&#37;C3&#37;A4res&#95;Karzinom&#95;und&#95;bili&#37;C3&#37;A4re&#95;Karzinome&#95;Langversion&#95;5.2.pdf</RefLink>
      </Reference>
      <Reference refNo="6">
        <RefAuthor>Klie JC</RefAuthor>
        <RefAuthor>Bugert M</RefAuthor>
        <RefAuthor>Boullosa B</RefAuthor>
        <RefAuthor>de Castilho RE</RefAuthor>
        <RefAuthor>Gurevych I</RefAuthor>
        <RefTitle>The INCEpTION platform: Machine-assisted and knowledge-oriented interactive annotation</RefTitle>
        <RefYear>2018</RefYear>
        <RefBookTitle>Proceedings of the 27th International Conference on Computational Linguistics: System Demonstrations; 2018 Aug 20-26; Santa Fe, New Mexico</RefBookTitle>
        <RefPage>5-9</RefPage>
        <RefTotal>Klie JC, Bugert M, Boullosa B, de Castilho RE, Gurevych I. The INCEpTION platform: Machine-assisted and knowledge-oriented interactive annotation. In: Proceedings of the 27th International Conference on Computational Linguistics: System Demonstrations; 2018 Aug 20-26; Santa Fe, New Mexico. Association for Computational Linguistics; 2018. p. 5-9.</RefTotal>
      </Reference>
      <Reference refNo="7">
        <RefAuthor>Gardazi NM</RefAuthor>
        <RefAuthor>Daud A</RefAuthor>
        <RefAuthor>Malik MK</RefAuthor>
        <RefAuthor>Bukhari A</RefAuthor>
        <RefAuthor>Alsahfi T</RefAuthor>
        <RefAuthor>Alshemaimri B</RefAuthor>
        <RefTitle>BERT applications in natural language processing: a review</RefTitle>
        <RefYear>2025</RefYear>
        <RefJournal>Artif Intell Rev</RefJournal>
        <RefPage>166</RefPage>
        <RefTotal>Gardazi NM, Daud A, Malik MK, Bukhari A, Alsahfi T, Alshemaimri B. BERT applications in natural language processing: a review. Artif Intell Rev. 2025;58(6):166. DOI: 10.1007&#47;s10462-025-11162-5</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1007&#47;s10462-025-11162-5</RefLink>
      </Reference>
      <Reference refNo="8">
        <RefAuthor>Saxena A</RefAuthor>
        <RefAuthor>Santhanavijayan A</RefAuthor>
        <RefTitle>A layer-wise survey on internal modifications in BERT and its variants: techniques, applications, and performance trade-offs</RefTitle>
        <RefYear>2026</RefYear>
        <RefJournal>International Journal of Data Science and Analytics</RefJournal>
        <RefPage>1</RefPage>
        <RefTotal>Saxena A, Santhanavijayan A. A layer-wise survey on internal modifications in BERT and its variants: techniques, applications, and performance trade-offs. International Journal of Data Science and Analytics. 2026;22(1):1. </RefTotal>
      </Reference>
      <Reference refNo="9">
        <RefAuthor>Joloudari JH</RefAuthor>
        <RefAuthor>Hussain S</RefAuthor>
        <RefAuthor>Nematollahi MA</RefAuthor>
        <RefAuthor>Bagheri R</RefAuthor>
        <RefAuthor>Fazl F</RefAuthor>
        <RefAuthor>Alizadehsani R</RefAuthor>
        <RefAuthor>Lashgari R</RefAuthor>
        <RefAuthor>Talukder A</RefAuthor>
        <RefTitle>BERT-deep CNN: State of the art for sentiment analysis of COVID-19 tweets</RefTitle>
        <RefYear>2023</RefYear>
        <RefJournal>Social Network Analysis and Mining</RefJournal>
        <RefPage>99</RefPage>
        <RefTotal>Joloudari JH, Hussain S, Nematollahi MA, Bagheri R, Fazl F, Alizadehsani R, Lashgari R, Talukder A. BERT-deep CNN: State of the art for sentiment analysis of COVID-19 tweets. Social Network Analysis and Mining. 2023;13(1):99. 
DOI: 10.1007&#47;s13278-023-01102-y</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1007&#47;s13278-023-01102-y</RefLink>
      </Reference>
      <Reference refNo="10">
        <RefAuthor>Chan B</RefAuthor>
        <RefAuthor>Schweter S</RefAuthor>
        <RefAuthor>M&#246;ller T</RefAuthor>
        <RefTitle>German&#8217;s next language model</RefTitle>
        <RefYear>2020</RefYear>
        <RefBookTitle>Proceedings of the 28th International Conference on Computational Linguistics; 2020 Dec 8-13; Barcelona, Spain (online)</RefBookTitle>
        <RefPage>6788-6796</RefPage>
        <RefTotal>Chan B, Schweter S, M&#246;ller T. German&#8217;s next language model. In: Scott D, Bel N, Zong C, editors. Proceedings of the 28th International Conference on Computational Linguistics; 2020 Dec 8-13; Barcelona, Spain (online). International Committee on Computational Linguistics; 2020. p. 6788-6796.</RefTotal>
      </Reference>
      <Reference refNo="11">
        <RefAuthor>Bressem KK</RefAuthor>
        <RefAuthor>Papaioannou JM</RefAuthor>
        <RefAuthor>Grundmann P</RefAuthor>
        <RefAuthor>Borchert F</RefAuthor>
        <RefAuthor>Adams LC</RefAuthor>
        <RefAuthor>Liu L</RefAuthor>
        <RefAuthor>Busch F</RefAuthor>
        <RefAuthor>Xu L</RefAuthor>
        <RefAuthor>Loyen JP</RefAuthor>
        <RefAuthor>Niehues SM</RefAuthor>
        <RefAuthor>Augustin M</RefAuthor>
        <RefAuthor>Grosser L</RefAuthor>
        <RefAuthor>Makowski MR</RefAuthor>
        <RefAuthor>Aerts HJWL</RefAuthor>
        <RefAuthor>L&#246;ser A</RefAuthor>
        <RefTitle>MedBERT.de: A comprehensive German BERT model for the medical domain &#91;preprint&#93;</RefTitle>
        <RefYear>2023</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefPage></RefPage>
        <RefTotal>Bressem KK, Papaioannou JM, Grundmann P, Borchert F, Adams LC, Liu L, Busch F, Xu L, Loyen JP, Niehues SM, Augustin M, Grosser L, Makowski MR, Aerts HJWL, L&#246;ser A. MedBERT.de: A comprehensive German BERT model for the medical domain &#91;preprint&#93;. arXiv. 2023. DOI: 10.48550&#47;arXiv.2303.08179</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.2303.08179</RefLink>
      </Reference>
      <Reference refNo="12">
        <RefAuthor>Conneau A</RefAuthor>
        <RefAuthor>Khandelwal K</RefAuthor>
        <RefAuthor>Goyal N</RefAuthor>
        <RefAuthor>Chaudhary V</RefAuthor>
        <RefAuthor>Wenzek G</RefAuthor>
        <RefAuthor>Guzm&#225;n F</RefAuthor>
        <RefAuthor>Grave E</RefAuthor>
        <RefAuthor>Ott M</RefAuthor>
        <RefAuthor>Zettlemoyer L</RefAuthor>
        <RefAuthor>Stoyanov V</RefAuthor>
        <RefTitle>Unsupervised cross-lingual representation learning at scale</RefTitle>
        <RefYear>2020</RefYear>
        <RefBookTitle></RefBookTitle>
        <RefPage>8440-8451</RefPage>
        <RefTotal>Conneau A, Khandelwal K, Goyal N, Chaudhary V, Wenzek G, Guzm&#225;n F, Grave E, Ott M, Zettlemoyer L, Stoyanov V. Unsupervised cross-lingual representation learning at scale. In: Jurafsky D, Chai J, Schluter N, Tetreault J, editors. Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics; 2020 Jul 5-10; online. Association for Computational Linguistics; 2020. p. 8440-8451.</RefTotal>
        <RefLink>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics; 2020 Jul 5-10; online</RefLink>
      </Reference>
      <Reference refNo="13">
        <RefAuthor>Grattafiori A</RefAuthor>
        <RefAuthor>Dubey A</RefAuthor>
        <RefAuthor>Jauhri A</RefAuthor>
        <RefAuthor>Pandey A</RefAuthor>
        <RefAuthor>Kadian A</RefAuthor>
        <RefAuthor>Al Dahle A</RefAuthor>
        <RefAuthor>Letman A</RefAuthor>
        <RefAuthor>Mathur A</RefAuthor>
        <RefAuthor>Schelten A</RefAuthor>
        <RefAuthor>Vaughan A</RefAuthor>
        <RefAuthor>Yang A</RefAuthor>
        <RefAuthor>Fan A</RefAuthor>
        <RefAuthor>Goyal A</RefAuthor>
        <RefAuthor>Hartshorn A</RefAuthor>
        <RefAuthor>Yang A</RefAuthor>
        <RefAuthor>Mitra A</RefAuthor>
        <RefAuthor>Sravankumar A</RefAuthor>
        <RefAuthor>Korenev A</RefAuthor>
        <RefAuthor>Hinsvark A</RefAuthor>
        <RefAuthor>Rao A</RefAuthor>
        <RefAuthor>Zhang A</RefAuthor>
        <RefAuthor>Rodriguez A</RefAuthor>
        <RefAuthor>Gregerson A</RefAuthor>
        <RefAuthor>Spataru A</RefAuthor>
        <RefAuthor>Roziere B</RefAuthor>
        <RefAuthor>Biron b</RefAuthor>
        <RefAuthor>Tang B</RefAuthor>
        <RefAuthor>Chern B</RefAuthor>
        <RefAuthor>Caucheteux C</RefAuthor>
        <RefAuthor>Nayak C</RefAuthor>
        <RefAuthor>Bi C</RefAuthor>
        <RefAuthor>Marra C</RefAuthor>
        <RefAuthor>McConnell C</RefAuthor>
        <RefAuthor>Keller C</RefAuthor>
        <RefAuthor>Touret C</RefAuthor>
        <RefAuthor>Wu C</RefAuthor>
        <RefAuthor>Wong C</RefAuthor>
        <RefAuthor>Ferrer CC</RefAuthor>
        <RefAuthor>Nikolaidis C</RefAuthor>
        <RefAuthor>Allonsius D</RefAuthor>
        <RefAuthor>Song D</RefAuthor>
        <RefAuthor>Pintz D</RefAuthor>
        <RefAuthor>Livshits D</RefAuthor>
        <RefAuthor>Wyatt D</RefAuthor>
        <RefAuthor>Esiobu D</RefAuthor>
        <RefAuthor>Choudhary D</RefAuthor>
        <RefAuthor>Mahajan D</RefAuthor>
        <RefAuthor>Garcia-Olano D</RefAuthor>
        <RefAuthor>Perino D</RefAuthor>
        <RefAuthor>Hupkes D</RefAuthor>
        <RefAuthor>Lakomkin E</RefAuthor>
        <RefAuthor>AlBadawy E</RefAuthor>
        <RefAuthor>Lobanova E</RefAuthor>
        <RefAuthor>Dinan E</RefAuthor>
        <RefAuthor>Smith EM</RefAuthor>
        <RefAuthor>Radenovic F</RefAuthor>
        <RefAuthor>Guzm&#225;n F</RefAuthor>
        <RefAuthor>Zhang F</RefAuthor>
        <RefAuthor>Synnaeve G</RefAuthor>
        <RefAuthor>Lee G</RefAuthor>
        <RefAuthor>Anderson GL</RefAuthor>
        <RefAuthor>Thattai G</RefAuthor>
        <RefAuthor>Nail G</RefAuthor>
        <RefAuthor>Mialon G</RefAuthor>
        <RefAuthor>Pang G</RefAuthor>
        <RefAuthor>Cucurell G</RefAuthor>
        <RefAuthor>Nguyen H</RefAuthor>
        <RefAuthor>Korevaar H</RefAuthor>
        <RefAuthor>Xu H</RefAuthor>
        <RefAuthor>Touvron H</RefAuthor>
        <RefAuthor>Zarov I</RefAuthor>
        <RefAuthor>Ibarra IA</RefAuthor>
        <RefAuthor>Kloumann I</RefAuthor>
        <RefAuthor>Misra I</RefAuthor>
        <RefAuthor>Evtimov I</RefAuthor>
        <RefAuthor>Zhang J</RefAuthor>
        <RefAuthor>Copet J</RefAuthor>
        <RefAuthor>Lee J</RefAuthor>
        <RefAuthor>GeffertJ</RefAuthor>
        <RefAuthor>Vranes J</RefAuthor>
        <RefAuthor>Park J</RefAuthor>
        <RefAuthor>Mahadeokar J</RefAuthor>
        <RefAuthor>Shah J</RefAuthor>
        <RefAuthor>van der Linde J</RefAuthor>
        <RefAuthor>Billock J</RefAuthor>
        <RefAuthor>Hong J</RefAuthor>
        <RefAuthor>Lee J</RefAuthor>
        <RefAuthor>Fu J</RefAuthor>
        <RefAuthor>Chi J</RefAuthor>
        <RefAuthor>Huang J</RefAuthor>
        <RefAuthor>Liu J</RefAuthor>
        <RefAuthor>Wang J</RefAuthor>
        <RefAuthor>Yu J</RefAuthor>
        <RefAuthor>Bitton J</RefAuthor>
        <RefAuthor>Spisak J</RefAuthor>
        <RefAuthor>Park J</RefAuthor>
        <RefAuthor>Rocca J</RefAuthor>
        <RefAuthor>Johnstun J</RefAuthor>
        <RefAuthor>Saxe J</RefAuthor>
        <RefAuthor>Jia J</RefAuthor>
        <RefAuthor></RefAuthor>
        <RefTitle>The llama 3 herd of models</RefTitle>
        <RefYear>2024</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefPage></RefPage>
        <RefTotal>Grattafiori A, Dubey A, Jauhri A, Pandey A, Kadian A, Al Dahle A, Letman A, Mathur A, Schelten A, Vaughan A, Yang A, Fan A, Goyal A, Hartshorn A, Yang A, Mitra A, Sravankumar A, Korenev A, Hinsvark A, Rao A, Zhang A, Rodriguez A, Gregerson A, Spataru A, Roziere B, Biron b, Tang B, Chern B, Caucheteux C, Nayak C, Bi C, Marra C, McConnell C, Keller C, Touret C, Wu C, Wong C, Ferrer CC, Nikolaidis C, Allonsius D, Song D, Pintz D, Livshits D, Wyatt D, Esiobu D, Choudhary D, Mahajan D, Garcia-Olano D, Perino D, Hupkes D, Lakomkin E, AlBadawy E, Lobanova E, Dinan E, Smith EM, Radenovic F, Guzm&#225;n F, Zhang F, Synnaeve G, Lee G, Anderson GL, Thattai G, Nail G, Mialon G, Pang G, Cucurell G, Nguyen H, Korevaar H, Xu H, Touvron H, Zarov I, Ibarra IA, Kloumann I, Misra I, Evtimov I, Zhang J, Copet J, Lee J, GeffertJ, Vranes J, Park J, Mahadeokar J, Shah J, van der Linde J, Billock J, Hong J, Lee J, Fu J, Chi J, Huang J, Liu J, Wang J, Yu J, Bitton J, Spisak J, Park J, Rocca J, Johnstun J, Saxe J, Jia J, et al. The llama 3 herd of models. arXiv. 2024. 
DOI: 10.48550&#47;arXiv.2407.21783</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.2407.21783</RefLink>
      </Reference>
      <Reference refNo="14">
        <RefAuthor>Bossy R</RefAuthor>
        <RefAuthor>Golik W</RefAuthor>
        <RefAuthor>Ratkovic Z</RefAuthor>
        <RefAuthor>Bessieres P</RefAuthor>
        <RefAuthor>N&#233;dellec C</RefAuthor>
        <RefTitle>BioNLP shared task 2013 &#8211; an overview of the bacteria biotope task</RefTitle>
        <RefYear>2013</RefYear>
        <RefBookTitle>Proceedings of the BioNLP shared task 2013 workshop; 2013 Aug 9; Sofia, Bulgaria</RefBookTitle>
        <RefPage>161-169</RefPage>
        <RefTotal>Bossy R, Golik W, Ratkovic Z, Bessieres P, N&#233;dellec C. BioNLP shared task 2013 &#8211; an overview of the bacteria biotope task. In: Proceedings of the BioNLP shared task 2013 workshop; 2013 Aug 9; Sofia, Bulgaria. Association for Computational Linguistics; 2013. p. 161-169.</RefTotal>
      </Reference>
      <Reference refNo="15">
        <RefAuthor>Wang Y</RefAuthor>
        <RefAuthor>Wang L</RefAuthor>
        <RefAuthor>Rastegar-Mojarad M</RefAuthor>
        <RefAuthor>Moon S</RefAuthor>
        <RefAuthor>Shen F</RefAuthor>
        <RefAuthor>Afzal N</RefAuthor>
        <RefAuthor>Liu S</RefAuthor>
        <RefAuthor>Zeng Y</RefAuthor>
        <RefAuthor>Mehrabi S</RefAuthor>
        <RefAuthor>Sohn S</RefAuthor>
        <RefAuthor>Liu H</RefAuthor>
        <RefTitle>Clinical information extraction applications: A literature review</RefTitle>
        <RefYear>2018</RefYear>
        <RefJournal>J Biomed Inform</RefJournal>
        <RefPage>34-49</RefPage>
        <RefTotal>Wang Y, Wang L, Rastegar-Mojarad M, Moon S, Shen F, Afzal N, Liu S, Zeng Y, Mehrabi S, Sohn S, Liu H. Clinical information extraction applications: A literature review. J Biomed Inform. 2018 Jan;77:34-49. DOI: 10.1016&#47;j.jbi.2017.11.011</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1016&#47;j.jbi.2017.11.011</RefLink>
      </Reference>
      <Reference refNo="16">
        <RefAuthor>Wu H</RefAuthor>
        <RefAuthor>Wang M</RefAuthor>
        <RefAuthor>Wu J</RefAuthor>
        <RefAuthor>Francis F</RefAuthor>
        <RefAuthor>Chang YH</RefAuthor>
        <RefAuthor>Shavick A</RefAuthor>
        <RefAuthor>Dong H</RefAuthor>
        <RefAuthor>Poon MTC</RefAuthor>
        <RefAuthor>Fitzpatrick N</RefAuthor>
        <RefAuthor>Levine AP</RefAuthor>
        <RefAuthor>Slater KT</RefAuthor>
        <RefAuthor>Handy A</RefAuthor>
        <RefAuthor>Karwath A</RefAuthor>
        <RefAuthor>Gkoutos GV</RefAuthor>
        <RefAuthor>Chelala C</RefAuthor>
        <RefAuthor>Shah AD</RefAuthor>
        <RefAuthor>Stewart R</RefAuthor>
        <RefAuthor>Collier N</RefAuthor>
        <RefAuthor>Alex B</RefAuthor>
        <RefAuthor>Whiteley W</RefAuthor>
        <RefAuthor>Sudlow C</RefAuthor>
        <RefAuthor>Roberts A</RefAuthor>
        <RefAuthor>Dobson RJB</RefAuthor>
        <RefTitle>A survey on clinical natural language processing in the United Kingdom from 2007 to 2022</RefTitle>
        <RefYear>2022</RefYear>
        <RefJournal>NPJ Digit Med</RefJournal>
        <RefPage>186</RefPage>
        <RefTotal>Wu H, Wang M, Wu J, Francis F, Chang YH, Shavick A, Dong H, Poon MTC, Fitzpatrick N, Levine AP, Slater KT, Handy A, Karwath A, Gkoutos GV, Chelala C, Shah AD, Stewart R, Collier N, Alex B, Whiteley W, Sudlow C, Roberts A, Dobson RJB. A survey on clinical natural language processing in the United Kingdom from 2007 to 2022. NPJ Digit Med. 2022 Dec 21;5(1):186. 
DOI: 10.1038&#47;s41746-022-00730-6</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1038&#47;s41746-022-00730-6</RefLink>
      </Reference>
      <Reference refNo="17">
        <RefAuthor>Yim WW</RefAuthor>
        <RefAuthor>Kwan SW</RefAuthor>
        <RefAuthor>Yetisgen M</RefAuthor>
        <RefTitle>Tumor reference resolution and characteristic extraction in radiology reports for liver cancer stage prediction</RefTitle>
        <RefYear>2016</RefYear>
        <RefJournal>J Biomed Inform</RefJournal>
        <RefPage>179-191</RefPage>
        <RefTotal>Yim WW, Kwan SW, Yetisgen M. Tumor reference resolution and characteristic extraction in radiology reports for liver cancer stage prediction. J Biomed Inform. 2016 Dec;64:179-191. 
DOI: 10.1016&#47;j.jbi.2016.10.005</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1016&#47;j.jbi.2016.10.005</RefLink>
      </Reference>
      <Reference refNo="18">
        <RefAuthor>Zhu S</RefAuthor>
        <RefAuthor>Gilbert M</RefAuthor>
        <RefAuthor>Ghanem AI</RefAuthor>
        <RefAuthor>Siddiqui F</RefAuthor>
        <RefAuthor>Thind K</RefAuthor>
        <RefTitle>Feasibility of using zero-shot learning in transformer-based natural language processing algorithm for key information extraction from head and neck tumor board notes &#91;ASTRO 2023 abstract&#93;</RefTitle>
        <RefYear>2023</RefYear>
        <RefJournal>Int J Radiat Oncol Biol Phys</RefJournal>
        <RefPage>e500</RefPage>
        <RefTotal>Zhu S, Gilbert M, Ghanem AI, Siddiqui F, Thind K. Feasibility of using zero-shot learning in transformer-based natural language processing algorithm for key information extraction from head and neck tumor board notes &#91;ASTRO 2023 abstract&#93;. Int J Radiat Oncol Biol Phys. 2023;117(2):e500.</RefTotal>
      </Reference>
      <Reference refNo="19">
        <RefAuthor>Hu Y</RefAuthor>
        <RefAuthor>Zuo X</RefAuthor>
        <RefAuthor>Zhou Y</RefAuthor>
        <RefAuthor>Peng X</RefAuthor>
        <RefAuthor>Huang J</RefAuthor>
        <RefAuthor>Keloth VK</RefAuthor>
        <RefAuthor>Zhang VJ</RefAuthor>
        <RefAuthor>Weng RL</RefAuthor>
        <RefAuthor>Shyr C</RefAuthor>
        <RefAuthor>Chen Q</RefAuthor>
        <RefAuthor>Jiang X</RefAuthor>
        <RefAuthor>Roberts KE</RefAuthor>
        <RefAuthor>Xu H</RefAuthor>
        <RefTitle>Information extraction from clinical notes: are we ready to switch to large language models&#63;</RefTitle>
        <RefYear>2026</RefYear>
        <RefJournal>J Am Med Inform Assoc</RefJournal>
        <RefPage>553-562</RefPage>
        <RefTotal>Hu Y, Zuo X, Zhou Y, Peng X, Huang J, Keloth VK, Zhang VJ, Weng RL, Shyr C, Chen Q, Jiang X, Roberts KE, Xu H. Information extraction from clinical notes: are we ready to switch to large language models&#63; J Am Med Inform Assoc. 2026 Mar 1;33(3):553-562. DOI: 10.1093&#47;jamia&#47;ocaf213</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1093&#47;jamia&#47;ocaf213</RefLink>
      </Reference>
      <Reference refNo="20">
        <RefAuthor>Li D</RefAuthor>
        <RefAuthor>Liang J</RefAuthor>
        <RefAuthor>Li W</RefAuthor>
        <RefAuthor>Wang X</RefAuthor>
        <RefAuthor>Cao L</RefAuthor>
        <RefAuthor>Yu K</RefAuthor>
        <RefTitle>CliCARE: Grounding large language models in clinical guidelines for decision support over longitudinal cancer electronic health records</RefTitle>
        <RefYear>2026</RefYear>
        <RefBookTitle>Proceedings of the 40th AAAI Conference on Artificial Intelligence; 2026 Jan 20-27; Singapore</RefBookTitle>
        <RefPage>31554-31562</RefPage>
        <RefTotal>Li D, Liang J, Li W, Wang X, Cao L, Yu K. CliCARE: Grounding large language models in clinical guidelines for decision support over longitudinal cancer electronic health records. In: Proceedings of the 40th AAAI Conference on Artificial Intelligence; 2026 Jan 20-27; Singapore. Washington, DC, USA: AAAI Press; 2026. 
p. 31554-31562.</RefTotal>
      </Reference>
      <Reference refNo="21">
        <RefAuthor>Berman E</RefAuthor>
        <RefAuthor>Sundberg Malek H</RefAuthor>
        <RefAuthor>Bitzer M</RefAuthor>
        <RefAuthor>Malek N</RefAuthor>
        <RefAuthor>Eickhoff C</RefAuthor>
        <RefTitle>Retrieval Augmented Therapy Suggestion for Molecular Tumor Boards: Algorithmic Development and Validation Study</RefTitle>
        <RefYear>2025</RefYear>
        <RefJournal>J Med Internet Res</RefJournal>
        <RefPage>e64364</RefPage>
        <RefTotal>Berman E, Sundberg Malek H, Bitzer M, Malek N, Eickhoff C. Retrieval Augmented Therapy Suggestion for Molecular Tumor Boards: Algorithmic Development and Validation Study. J Med Internet Res. 2025 Mar 5;27:e64364. DOI: 10.2196&#47;64364</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.2196&#47;64364</RefLink>
      </Reference>
    </References>
    <Media>
      <Tables>
        <Table format="png">
          <MediaNo>1</MediaNo>
          <MediaID>1</MediaID>
          <Caption><Pgraph><Mark1>Table 1: Overview of the F1 score evaluation results for each parameter, based on the XLM-RoBERTa, RegEx and LLaMA approaches. </Mark1><LineBreak></LineBreak>The following are depicted: The support of the parameter within the test set (&#35;), as well as the <Mark2>lenient</Mark2> (L), <Mark2>strict</Mark2> (S) and <Mark2>manual</Mark2> (M) evaluation results. Grey highlighted rows mark the 14 parameters we focused on. Although no evaluation results were yet available for Child-Pugh and LiMAx, these were included to provide a comprehensive overview of the parameters.</Pgraph></Caption>
        </Table>
        <NoOfTables>1</NoOfTables>
      </Tables>
      <Figures>
        <Figure width="1033" height="574" format="png">
          <MediaNo>1</MediaNo>
          <MediaID>1</MediaID>
          <Caption><Pgraph><Mark1>Figure 1: The aggregated micro-F1 scores of BERT, XLM-RoBERTa and MedBERT on the test set are compared across sentence- and document-level extraction using </Mark1><Mark1><Mark2>lenient</Mark2></Mark1><Mark1> evaluation.</Mark1></Pgraph></Caption>
        </Figure>
        <NoOfPictures>1</NoOfPictures>
      </Figures>
      <InlineFigures>
        <NoOfPictures>0</NoOfPictures>
      </InlineFigures>
      <Attachments>
        <NoOfAttachments>0</NoOfAttachments>
      </Attachments>
    </Media>
  </OrigData>
</GmsArticle>