<?xml version="1.0" encoding="iso-8859-1" standalone="no"?>
<!DOCTYPE GmsArticle SYSTEM "http://www.egms.de/dtd/2.0.34/GmsArticle.dtd">
<GmsArticle xmlns:xlink="http://www.w3.org/1999/xlink">
  <MetaData>
    <Identifier>mibe000313</Identifier>
    <IdentifierDoi>10.3205/mibe000313</IdentifierDoi>
    <IdentifierUrn>urn:nbn:de:0183-mibe0003134</IdentifierUrn>
    <ArticleType>Research Article</ArticleType>
    <TitleGroup>
      <Title language="en">Semantic retrieval-augmented translation for radiology reports using small language models: Algorithm development and evaluation</Title>
      <TitleTranslated language="de">Semantische Retrieval-gest&#252;tzte &#220;bersetzung von Radiologiebefunden mit kleinen Sprachmodellen: Algorithmusentwicklung und -evaluation</TitleTranslated>
    </TitleGroup>
    <CreatorList>
      <Creator>
        <PersonNames>
          <Lastname>Reichenpfader</Lastname>
          <LastnameHeading>Reichenpfader</LastnameHeading>
          <Firstname>Daniel</Firstname>
          <Initials>D</Initials>
          <AcademicTitle>Prof.</AcademicTitle>
        </PersonNames>
        <Address>Berner Fachhochschule, Technik und Informatik, Forschung und Dienstleistung, Quellgasse 21, 2502 Biel&#47;Bienne, Switzerland, Phone: &#43;41 31 848 60 93<Affiliation>Faculty of Medicine, University of Geneva, Switzerland</Affiliation><Affiliation>Institute for Patient-Centered Digital Health, Bern University of Applied Sciences, Bern, Switzerland</Affiliation></Address>
        <Email>daniel.reichenpfader&#64;bfh.ch</Email>
        <Creatorrole corresponding="yes" presenting="no">author</Creatorrole>
      </Creator>
      <Creator>
        <PersonNames>
          <Lastname>Dennst&#228;dt</Lastname>
          <LastnameHeading>Dennst&#228;dt</LastnameHeading>
          <Firstname>Fabio</Firstname>
          <Initials>F</Initials>
        </PersonNames>
        <Address>
          <Affiliation>Department of Radiation Oncology, Bern University Hospital, University of Bern, Switzerland</Affiliation>
        </Address>
        <Email>fabio.dennstaedt&#64;insel.ch</Email>
        <Creatorrole corresponding="no" presenting="no">author</Creatorrole>
      </Creator>
    </CreatorList>
    <PublisherList>
      <Publisher>
        <Corporation>
          <Corporatename>German Medical Science GMS Publishing House</Corporatename>
        </Corporation>
        <Address>D&#252;sseldorf</Address>
      </Publisher>
    </PublisherList>
    <SubjectGroup>
      <SubjectheadingDDB>610</SubjectheadingDDB>
      <Keyword language="en">machine translation</Keyword>
      <Keyword language="en">few-shot&#47;zero-shot MT</Keyword>
      <Keyword language="en">human evaluation</Keyword>
      <Keyword language="en">domain adaptation</Keyword>
      <Keyword language="en">healthcare applications</Keyword>
      <Keyword language="en">natural language processing</Keyword>
      <Keyword language="en">clinical NLP</Keyword>
      <Keyword language="en">NLP in resource-constrained settings</Keyword>
      <Keyword language="de">maschinelle Few-Shot-&#47;Zero-Shot-&#220;bersetzung</Keyword>
      <Keyword language="de">Evaluation</Keyword>
      <Keyword language="de">nat&#252;rliche Sprachverarbeitung</Keyword>
      <Keyword language="de">Anwendungen im Gesundheitswesen</Keyword>
      <Keyword language="de">Computerlinguistik</Keyword>
      <SectionHeading language="en">ISCB GMDS 2026</SectionHeading>
    </SubjectGroup>
    <DatePublishedList>
      <DatePublished>20260922</DatePublished>
    </DatePublishedList>
    <Language>engl</Language>
    <License license-type="open-access" xlink:href="http://creativecommons.org/licenses/by/4.0/">
      <AltText language="en">This is an Open Access article distributed under the terms of the Creative Commons Attribution 4.0 License.</AltText>
      <AltText language="de">Dieser Artikel ist ein Open-Access-Artikel und steht unter den Lizenzbedingungen der Creative Commons Attribution 4.0 License (Namensnennung).</AltText>
    </License>
    <SourceGroup>
      <Journal>
        <ISSN>1860-9171</ISSN>
        <Volume>22</Volume>
        <JournalTitle>GMS Medizinische Informatik, Biometrie und Epidemiologie</JournalTitle>
        <JournalTitleAbbr>GMS Med Inform Biom Epidemiol</JournalTitleAbbr>
      </Journal>
    </SourceGroup>
    <ArticleNo>15</ArticleNo>
  </MetaData>
  <OrigData>
    <Abstract language="de" linked="yes"><Pgraph><Mark1>Einleitung:</Mark1> Eine pr&#228;zise maschinelle &#220;bersetzung (MT) klinischer Texte ist entscheidend f&#252;r die mehrsprachige Patientenversorgung, den grenz&#252;berschreitenden Datenaustausch und die Reproduzier-barkeit von Forschung. Der Einsatz kommerzieller APIs f&#252;r solche Anwendungen wirft jedoch erhebliche Datenschutzbedenken auf. Wir stellen Semantic Retrieval-Augmented Translation (S-RAT) vor, ein leichtgewichtiges, ontologiegest&#252;tztes Framework, das die Konzeptextraktion aus dem Unified Medical Language System (UMLS) und SNOMED CT in Prompts f&#252;r gro&#223;e Sprachmodelle integriert, um die terminologische Genauigkeit ohne Fine-Tuning oder externe APIs zu verbessern.</Pgraph><Pgraph><Mark1>Methoden:</Mark1> Auf Basis von 500 englischsprachigen radiologischen Befunden aus MIMIC-CXR evaluierten wir mehrere S-RAT-Varianten unter Verwendung quelloffener Sprachmodelle (gemma3, med42-v2, deepseek-r1 und gpt-oss) sowie eines kommerziellen Frontier-Modells (<TextGroup><PlainText>Gemini</PlainText></TextGroup> 3 Pro). Wir verglichen referenzfreie Qualit&#228;tsmetriken (COMET-src und GEMBA-DA-noref) mit menschlichen Annotationen. Zus&#228;tzlich untersuchten wir die Abdeckung der deutschen SNOMED-CT-Edition f&#252;r den spezifischen Anwendungsfall der &#220;bersetzung radiologischer Befunde.</Pgraph><Pgraph><Mark1>Ergebnisse:</Mark1> S-RAT-Prompting zeigte keinen konsistenten Vorteil gegen&#252;ber Standard-Prompting. COMET-src korrelierte nur schwach mit menschlichen Bewertungen der inhaltlichen Angemessenheit (&#961;&#8776;0,14, p&#8776;0,05), w&#228;hrend GEMBA-DA-noref keine relevante Korrelation aufwies. Die deutsche SNOMED-CT-Edition deckte lediglich 32,6&#37; der extrahierten Konzepte ab.</Pgraph><Pgraph><Mark1>Diskussion:</Mark1> Die Ergebnisse verdeutlichen die aktuellen Einschr&#228;nkungen mehrsprachiger Ontologieressourcen sowie die begrenzte Eignung allgemeiner MT-Evaluationsmetriken im klinischen Kontext. S-RAT demonstriert die Machbarkeit einer datenschutzkonformen, lokal einsetzbaren und ontologiegest&#252;tzten &#220;bersetzung und bildet eine Grundlage f&#252;r zuk&#252;nftige Arbeiten zur dom&#228;nenspezifischen Evaluation, zur Erweiterung von Ontologien sowie zum lokalen Einsatz von Sprachmodellen im Gesundheitswesen.</Pgraph></Abstract>
    <Abstract language="en" linked="yes"><Pgraph><Mark1>Introduction:</Mark1> Accurate machine translation (MT) of clinical text is essential for multilingual patient care, cross-border data sharing, and research reproducibility, yet the use of commercial APIs for such tasks raises substantial privacy concerns. We present Semantic Retrieval-Augmented Translation (S-RAT), a lightweight, ontology-guided framework that integrates the Unified Medical Language System (UMLS) and SNOMED CT concept retrieval into large language model prompts to improve medical terminology fidelity without fine-tuning or external APIs.</Pgraph><Pgraph><Mark1>Methods:</Mark1> Using 500 English radiology reports from MIMIC-CXR, we evaluated multiple S-RAT variants with open-source language models (gemma3, med42-v2, deepseek-r1, and gpt-oss) as well as a commercial frontier model (Gemini 3 Pro). We compared reference-free quality estimation metrics (COMET-src and GEMBA-DA-noref) with human annotation. Furthermore, we assessed the translation coverage of the German SNOMED CT edition for the specific use case of radiology report translation.</Pgraph><Pgraph><Mark1>Results:</Mark1> S-RAT prompting did not consistently outperform standard prompting, and COMET-src correlated only weakly (&#961;&#8776;0.14, p&#8776;0.05) with human adequacy judgments, while GEMBA-DA-noref showed no meaningful correlation. The German SNOMED CT edition covered only 32.6&#37; of extracted concepts.</Pgraph><Pgraph><Mark1>Discussion:</Mark1> These findings highlight the current limitations of multilingual ontology resources and the application of general-domain MT metrics in clinical contexts. S-RAT demonstrates the feasibility of privacy-pre<TextGroup><PlainText>s</PlainText></TextGroup>erving, locally deployable, and ontology-augmented translation and provides a foundation for future work on domain-specific evaluation, ontology enrichment, and local deployment of language models in healthcare settings.</Pgraph></Abstract>
    <TextBlock name="1 Introduction" linked="yes">
      <MainHeadline>1 Introduction</MainHeadline><Pgraph>Large language models (LLMs) have demonstrated strong capabilities in few-shot and zero-shot machine translation (MT), including in the biomedical domain <TextLink reference="1"></TextLink>. In clinical practice, accurate MT of free-text documents such as radiology reports is essential for cross-lingual collaboration, international patient care, and dataset interoperability. In research, multilingual datasets are necessary for comparing the performance of developed algorithms across languages. However, in many countries, regulatory and privacy concerns prevent the submission of protected health information to commercial APIs, thereby limiting the applicability of leading proprietary LLMs in real-world clinical use cases. Open-source models such as gemma3 <TextLink reference="2"></TextLink>, gpt-oss <TextLink reference="3"></TextLink>, and Apertus <TextLink reference="4"></TextLink> have been rapidly improving in performance, and for single-turn translation tasks, model size is less critical than in dialogue or reasoning-heavy settings. This creates an opportunity for smaller, locally deployable models (&#60;15 billion parameters) in hospital infrastructures, but also raises questions about how effectively such models can be guided using structured external knowledge.</Pgraph><Pgraph>Existing biomedical MT approaches rely on either large proprietary models, which raise privacy concerns, or open models fine-tuned with large annotated corpora. However, neither approach leverages structured clinical ontologies for translation guidance. We propose a method called semantic retrieval-augmented translation (S-RAT), which identifies Unified Medical Language System (UMLS) <TextLink reference="5"></TextLink> concepts in the source text, retrieves the corresponding SNOMED CT preferred terms in the target language, and injects these translations directly into the prompt. This aims at guiding the LLM in accurate translation of clinical terms without requiring fine-tuning or cloud APIs. Unlike standard retrieval-augmented generation (RAG), which retrieves unstructured text, S-RAT performs structured retrieval from the SNOMED CT ontology, mapping source-language medical entities to standardized target-language terms. This structured retrieval reduces ambiguity and ensures terminological consistency, particularly important in clinical subdomains such as radiology.</Pgraph><Pgraph>S-RAT is designed to be lightweight, privacy-preserving, and reproducible. With this paper, we evaluate whether the German national SNOMED CT edition contains sufficient translated concepts for the radiology domain, and whether retrieval-augmented prompting affects translation accuracy compared with standard prompting. We further assess whether COMET-src, a reference-free MT quality estimation metric, aligns with human expert judgments. Specifically, we test the following hypotheses:</Pgraph><Pgraph><UnorderedList><ListItem level="1">H1: Translation coverage exceeds 80&#37;, defined as the proportion of extracted concept instances for which a German translation could be retrieved.</ListItem><ListItem level="1">H2: Retrieval-augmented translation with S-RAT using a small open-source LLM (gemma3:4b) differs from standard prompting.</ListItem><ListItem level="1">H3: The two general-domain quality estimation metrics COMET-src and GEMBA-DA-noref correlate at most weakly (&#961;&#60;0.3) with human judgment on radiology report translations.</ListItem></UnorderedList></Pgraph><SubHeadline>1.1 Related work</SubHeadline><Pgraph>In the biomedical domain, both MT and LLMs have been evaluated extensively. A recent study comparing GPT-4, GPT-3.5 and Qwen1.5 found that GPT-4 provided the most accurate translations of CT and MRI reports across nine languages <TextLink reference="6"></TextLink>. Similarly, a comparative analysis of ChatGPT (GPT-4) and Google Translate on patient discharge instructions reported higher accuracy of ChatGPT <TextLink reference="1"></TextLink>. To support such evaluations and facilitate the traceability of results, various multilingual corpora have been introduced. The MIMIC-CXR dataset provides large-scale radiology reports but only in English. The OpenWHO dataset comprises 26,824 sentences across more than 20 languages <TextLink reference="7"></TextLink>, the Multilingual Medical Corpus offers a large collection of biomedical texts across English, Spanish, French, and Italian <TextLink reference="8"></TextLink>, and MedEV contains approximately 360,000 Vietnamese&#8211;English sentence pairs for medical MT <TextLink reference="9"></TextLink>. Despite these advances, the availability of open multilingual clinical datasets, especially for radiology, remains limited.</Pgraph><Pgraph>Beyond datasets, several strategies have been proposed to adapt LLMs for medical MT. Rios fine-tuned LLMs with medical glossaries via QLoRA, achieving improved MT evaluation scores with higher terminology accuracy <TextLink reference="10"></TextLink>. RAG has also emerged as a promising technique to enhance translation by injecting relevant knowledge at inference time <TextLink reference="11"></TextLink>. RAGtrans introduced a benchmark of 79,000 knowledge-intensive sentences and demonstrated that retrieval-augmented translation with unstructured documents can outperform instruction-tuning alone <TextLink reference="12"></TextLink>.</Pgraph><Pgraph>In contrast to these approaches, our method does not rely on large fine-tuning datasets or unstructured document retrieval. Instead, we leverage structured clinical ontologies: SNOMED CT concept translations are directly retrieved and incorporated into prompts, upgrading a small, locally deployed LLM with improved terminology coverage in a privacy-compliant way.</Pgraph></TextBlock>
    <TextBlock name="2 Methods" linked="yes">
      <MainHeadline>2 Methods</MainHeadline><Pgraph>In this section, we provide an overview of the complete S-RAT pipeline as well as introduce the applied datasets, models, tools, and evaluation metrics.</Pgraph><SubHeadline>2.1 S-RAT pipeline</SubHeadline><Pgraph>To augment LLM prompts with quality-assured translations of extracted medical terms, we implement a translation retrieval pipeline: Figure 1 <ImgLink imgNo="1" imgType="figure" /> summarizes the complete S-RAT workflow. Starting from an English radiology report, clinical concepts are extracted and normalized to SNOMED CT identifiers, translated using the German SNOMED CT edition (or a fallback dictionary if necessary), and finally injected into the translation prompt before the LLM generates the German report. The figure illustrates where structured terminology retrieval augments an ot<TextGroup><PlainText>h</PlainText></TextGroup>erwise standard prompting workflow. Licenses and tools are detailed in Appendix A (Attachment 1 <AttachmentLink attachmentNo="1" />).</Pgraph><Pgraph><OrderedList><ListItem level="1" levelPosition="1" numString="1."><Mark1>Concept extraction:</Mark1> Clinical concepts are identified and extracted using a locally deployed commercial service (Azure Text Analytics for health) <TextLink reference="13"></TextLink>.</ListItem><ListItem level="1" levelPosition="2" numString="2."><Mark1>Normalization:</Mark1> UMLS Concept Unique Identifiers (CUIs) and, if available, the SNOMED CT identifier (SCTID) are obtained together with the extracted concepts. If the SCTID is not provided, it is obtained via the UMLS REST API <TextLink reference="14"></TextLink>.</ListItem><ListItem level="1" levelPosition="3" numString="3."><Mark1>Multilingual translation:</Mark1> Verified translations of the identified concepts are retrieved from a locally deployed SNOMED CT terminology server (Snowstorm), which is populated with the respective national edition of SNOMED CT (in this case, German) <TextLink reference="15"></TextLink>.</ListItem><ListItem level="1" levelPosition="4" numString="4."><Mark1>Fallback translation:</Mark1> If a translation is not found in the national edition of SNOMED CT, the system attempts to retrieve the translation from a manually populated list of human-verified translations of common concepts.</ListItem><ListItem level="1" levelPosition="5" numString="5."><Mark1>Prompt construction:</Mark1> The translated terms are included within the user prompt.</ListItem></OrderedList></Pgraph><Pgraph><ImgPlaceholder imgNo="1" imgType="figure"/></Pgraph><SubHeadline>2.2 Datasets, models, and prompts</SubHeadline><Pgraph>We evaluate our method using the MIMIC-CXR dataset, which contains 227,835 imaging studies including reports for 65,379 patients presenting to the Beth Israel Deaconess Medical Center Emergency Department between 2011 and 2016 <TextLink reference="16"></TextLink>. For our analysis, we randomly sampled 500 reports. In addition, we selected two reports manually for in-context learning. Details on data preprocessing and sample characteristics are provided in Appendix B (Attachment 1 <AttachmentLink attachmentNo="1" />).</Pgraph><Pgraph>Our experiments are conducted with open-source models that can be realistically deployed locally on institutional infrastructure, while still being large enough to perform MT. Therefore, we chose model sizes between four and twelve billion parameters and focused on the gemma3 family, a suite of state-of-the-art, medium-sized open-source generative models with strong multilingual capabilities <TextLink reference="2"></TextLink>. The smallest 1b variant was excluded after preliminary tests showed poor overall performance, which resulted in using the 4b and 12b model variants. We also include med42-v2:8b, a fine-tuned model for the medical domain, based on llama3.3 <TextLink reference="17"></TextLink> as well as deepseek-r1:8b <TextLink reference="18"></TextLink>. As state-of-the-art baseline, we use gpt-oss:120b, the strongest open-source model available at the time of writing <TextLink reference="3"></TextLink>. Additionally, we use Gemini 3 Pro as state-of-the-art frontier model <TextLink reference="19"></TextLink>.</Pgraph><Pgraph>We base our initial system and user prompts on the work of Chen et al. <TextLink reference="20"></TextLink>, who evaluated various MT tools for translating critical care-related content, building on an earlier industry report. We adapt their prompt to the use case of radiology report translation. To enhance the concept translation module, we further developed a more sophisticated system prompt. This refined version incorporates additional instructions and leverages a two-shot in-context learning setup with two manually translated reports, used in the S-RAT-Shot and S-RAT-Curated variants. Prior work by Brown et al. has shown that two examples can already substantially improve performance <TextLink reference="21"></TextLink>. The full set of prompts is provided in Appendix G (Attachment 1 <AttachmentLink attachmentNo="1" />). All resources will be made publicly available via Zenodo <TextLink reference="22"></TextLink>.</Pgraph><SubHeadline>2.3 Evaluation metrics</SubHeadline><Pgraph>MT evaluation has traditionally been relying on reference-based metrics, which compare system outputs against human-produced reference translations. Examples for reference-based metrics are BLEU <TextLink reference="23"></TextLink>, ROUGE <TextLink reference="24"></TextLink>, METEOR <TextLink reference="25"></TextLink>, and BERTScore <TextLink reference="26"></TextLink>. One of the major disadvantages besides low alignment with human preferences as shown by <TextLink reference="27"></TextLink> for clinical text summarization is the need for human-generated reference summaries to evaluate against. This issue is aggravated in the healthcare domain, where physicians and other healthcare professionals have only limited resources for annotation.</Pgraph><Pgraph>Recently, research in the domain of quality estimation (QE) has shifted toward developing and validating reference-free metrics, which do not require reference translations and instead assess the adequacy and fluency of the output directly <TextLink reference="28"></TextLink>. Examples for reference-free metrics are YiSi-2 <TextLink reference="29"></TextLink>, Prism-src <TextLink reference="30"></TextLink>, SentSim <TextLink reference="31"></TextLink>, RefFreeEval <TextLink reference="32"></TextLink>, and MT-Ranker <TextLink reference="33"></TextLink>. Another trend also explores the use of LLMs as automatic evaluators, also called &#8220;LLM-as-a-judge&#8221; <TextLink reference="34"></TextLink> In the domain of MT evaluation, an LLM can be prompted to perform QE. While such approaches show strong correlation with human judgments, they are sensitive to prompt design and can exhibit systematic biases <TextLink reference="35"></TextLink>, <TextLink reference="36"></TextLink>.</Pgraph><Pgraph>In this paper, we assess the two QE metrics COMET-src and GEMBA-DA-noref. COMET-src is a reference-free variant of the widely used COMET metric. Both COMET and COMET-src have been shown to be among the best-performing MT evaluation metrics <TextLink reference="37"></TextLink>. A specific COMET variant, CometKiwi-DA-XL <TextLink reference="38"></TextLink>, is used for the automated evaluation of the General Machine Translation Task at the 2024 Conference on Machine Translation (WMT) <TextLink reference="39"></TextLink>. For our experiments, we use the largest available model variant of COMET-src, Unbabel&#47;wmt23-cometkiwi-da-xxl, comprising 10.5 billion parameters and requiring a minimum of 44 GB of GPU memory <TextLink reference="40"></TextLink>.</Pgraph><Pgraph>GEMBA (GPT Estimation Metric Based Assessment) is an LLM-based evaluation approach that uses prompting to directly estimate translation quality. In its reference-free variant, GEMBA-DA-NoRef prompts the model to assign a direct assessment score between 0 and 100 without access to a reference translation and has shown strong system-level correlation with human judgments in prior evaluations on general-domain benchmarks <TextLink reference="41"></TextLink>. Related approaches such as EAPrompt extend this idea by incorporating explicit error-aware prompting to obtain more fine-grained quality estimates, however at the cost of increased prompting complexity <TextLink reference="42"></TextLink>.</Pgraph></TextBlock>
    <TextBlock name="3 Results" linked="yes">
      <MainHeadline>3 Results</MainHeadline><SubHeadline>3.1 Testing variants of the S-RAT pipeline</SubHeadline><Pgraph>We tested four variants of the S-RAT pipeline (Base, Dict, Shot, and Curated). Each variant was evaluated on the same dataset with both gemma3 model variants (4b and 12b). Furthermore, the best-performing variant (Curated) was evaluated using two additional models (med42-v2:8b, deepseek-r1:8b). Additionally, we evaluated S-RAT-Base on gpt-oss:120b as large open-source model and Gemini 3 Pro as state-of-the-art commercial model, resulting in a total of twelve configurations.</Pgraph><Pgraph>S-RAT-Base uses the basic system prompt (Appendix G in Attachment 1 <AttachmentLink attachmentNo="1" />) and includes all previously described S-RAT components, except the fallback translation list.</Pgraph><Pgraph> S-RAT-Dict extends the base pipeline by adding the fallback translation list (generation described in Appendix E in Attachment 1 <AttachmentLink attachmentNo="1" />).</Pgraph><Pgraph>S-RAT-Shot applies a more complex system prompt, containing two manually curated translation examples for in-context learning.</Pgraph><Pgraph>S-RAT-Curated incorporates a manually curated dictionary of 173 concepts. This dictionary combines: (i) automatically translated and verified terms, (ii) entries from the fallback list (S-RAT-Dict), and (iii) frequent uni-, bi-, and trigrams from the dataset. In this variant, concept translations are added to the S-RAT-Shot prompt, but only for exact string matches.</Pgraph><Pgraph>For each configuration, the pipeline was applied to the full subset of 500 reports. Each report was translated based on two variants: once with keywords included in the prompt (-K, &#8220;S-RAT&#8221;) and once without (-NK, &#8220;standard&#8221;), yielding 1,000 translations per configuration.</Pgraph><Pgraph>We then computed the COMET-src score for every translation. ranging from 0 to 1 with higher values reflecting better translation quality. To evaluate whether our S-RAT approach differs from the standard approach, we applied a two-step evaluation strategy. For the primary analysis, we collapsed across model configurations: for each of the 500 clinical reports, we computed the mean COMET score across the eight models separately for the standard and S-RAT translation variants. These paired values were then compared using a Wilcoxon signed-rank test, providing an overall assessment of whether the S-RAT variant achieved higher translation quality.</Pgraph><Pgraph>For the secondary analysis, we compared the S-RAT and standard variants within each of the twelve model configurations using Wilcoxon signed-rank tests. Holm correction was applied to adjust for multiple comparisons. Effect sizes were reported as rank-based r. Together, these analyses tested both the overall benefit of the S-RAT variant and its consistency across different model architectures.</Pgraph><Pgraph><Mark1>Result:</Mark1> Keyword-enhanced translation does not perform consistently better than standard prompting. In Table 1 <ImgLink imgNo="1" imgType="table" />, we report the performance of each configuration. In the primary analysis, the keyword variant did not significantly outperform the standard variant (Wilcoxon signed-rank test, W&#61;62563.00, p&#61;9.847e-01, r&#61;0.001, n&#61;500) averaged across all model configurations. In the secondary analysis, three variant comparisons reached significance after Holm correction. Baseline performance (gpt-oss:120b) decreased significantly using keyword-enhanced translation. Gemini 3 Pro did not outperform gemma3:12b, but its performance increased significantly using keyword-enhanced translation. The performance of the fine-tuned model med42-v2:8b showed overall low performance and significantly decreased using keyword-enhanced translation. Median differences favored the keyword-enhanced variant in seven out of twelve comparisons, see Appendix D (Attachment 1 <AttachmentLink attachmentNo="1" />).</Pgraph><SubHeadline>3.2 Verifying SNOMED CT as feasible resource for concept translation</SubHeadline><Pgraph>To verify whether SNOMED CT is a feasible resource for concept translation, we compute the observed translation coverage of our pipeline. We analyze the results of the S-RAT-Base configuration and obtain the total number of extracted concept instances (<Mark2>T</Mark2>). We then determine the number of instances that could not be translated (<Mark2>U</Mark2>) based on the list of untranslated concepts and their frequencies.</Pgraph><Pgraph>Our primary metric, the observed translation coverage, is calculated as:</Pgraph><Pgraph><Indentation><ImgLink imgNo="1" imgType="inlineFigure" /> </Indentation></Pgraph><Pgraph>We report this proportion with a 95&#37; confidence interval and perform a one-sided exact binomial test to assess whether coverage exceeds the predefined 80&#37; feasibility threshold.</Pgraph><Pgraph>To account for vocabulary breadth, we additionally compute distinct-type coverage, defined as the proportion of unique extracted concept types for which a German translation was available. We also present the distribution of untranslated concepts by frequency (Pareto analysis), highlighting whether some terms account for most untranslated instances.</Pgraph><Pgraph>This analysis measures only the proportion of extracted concepts covered by SNOMED CT. We do not evaluate the correctness of concept mappings or the clinical adequacy of translations due to the absence of human annotation. Results are therefore interpreted as coverage conditional on the output of the fixed extraction system. However, we perform a separate ablation study to assess recall and precision of the commercial entity recognition tool.</Pgraph><Pgraph><Mark1>Result:</Mark1> The German SNOMED CT extension provides insufficient coverage for radiological use cases. A total of 6,305 concept instances were extracted from 500 radiology reports. Of these, 4,247 instances could not be translated into German, corresponding to an observed translation coverage of 32.64&#37; (95&#37; CI: 31.5&#8211;33.8&#37;). This coverage falls well below the predefined feasibility threshold of 80&#37;. Compared to S-RAT-Base, S-RAT-Dict improved translation coverage from 32.64&#37; to 81.65&#37;. S-RAT-Curated nearly quadrupled the average number of extracted concepts per report from 12.61 to 48.35. The most frequent untranslated terms included Chest (n&#61;375), Consolidation (n&#61;180), Male population group (n&#61;155), and Abnormally opaque structure (n&#61;150). A Pareto analysis showed that the top 20 untranslated terms accounted for the entire set of untranslated instances, with the ten most common terms already contributing over 70&#37;. At the vocabulary level, 593 unique concept types lacked a German translation, resulting in a distinct-type coverage of 77.6&#37;. This suggests that while most unique concept types were represented, frequent radiology terms remained systematically untranslated, disproportionately reducing instance-level coverage. Taken together, these results indicate that the German SNOMED CT edition, in its current form, does not provide sufficient coverage for translating radiology-related concepts from English to German.</Pgraph><SubHeadline>3.3 Correlating automated QE metrics with human preferences</SubHeadline><Pgraph>We base our human annotation methodology on the work of Chatzikoumi <TextLink reference="43"></TextLink> One co-author, a radio-oncologist and therefore domain expert, performs a blinded assessment of the two translation variants (-K vs. -NK) based on the results of the S-RAT-Curated-4b configuration (n&#61;199). For evaluation, the order of each triplet is randomly shuffled. Each triplet is comparatively rated for accuracy (&#61; adequacy) using a bipolar scale ranging from &#8211;5 (strong preference for the -NK translation) to &#43;5 (strong preference for the -K translation), based on existing work of Van Veen et al., who applied this approach for comparison of human- and machine-generated summaries of clinical texts <TextLink reference="27"></TextLink>.</Pgraph><Pgraph>To compare these human preferences with each of the two automated metrics, we compute a difference score for each triplet by subtracting the QE score of the standard translation from that of the S-RAT translation, so that positive values indicate a higher QE score. Human ratings are treated as signed preference strengths. This allows direct comparison between human and metric scores: both are positive when S-RAT is preferred and negative when the standard translation is preferred.</Pgraph><Pgraph>To assess the degree to which COMET-src and GEMBA-DA-noref differences align with human preferences, we compute Spearman&#8217;s rank correlation coefficient (Spearman&#8217;s &#961;) between the two vectors. In addition, we report the p-value associated with the correlation to determine whether the observed association is statistically significant.</Pgraph><Pgraph><Mark1>Result:</Mark1> Automated metrics show limited alignment with human evaluation scores. The expert annotations (n&#61;199) revealed a distribution where 47.2&#37; of pairs were rated as equivalent, 31.2&#37; favored S-RAT translations, and 21.6&#37; favored standard translations. The mean rating of 0.231 (SD&#61;1.399) indicates a slight overall preference for S-RAT. Spearman&#8217;s rank correlation analysis revealed that COMET-src differences showed a weak but statistically significant correlation with human preferences (Spearman&#8217;s &#961;&#61;0.139, p&#61;0.0499, n&#61;199), see Figure 2 <ImgLink imgNo="2" imgType="figure" />: Each point corresponds to one translated report pair, with the horizontal axis showing the COMET-src score difference (S-RAT minus standard prompting) and the vertical axis showing the corresponding human preference rating. In contrast, GEMBA-DA-noref differences showed no meaningful correlation (Spearman&#8217;s &#961;&#61;0.007, p&#61;0.917, n&#61;198), despite its prior performance on general-domain MT benchmarks. While COMET-src captures a small fraction of the variance in expert adequacy judgments, GEMBA-DA-noref fails to align with human preferences at all. Given that both measures are intended to reflect translation quality, stronger correspondence would be expected if the metrics adequately modeled domain-specific adequacy. The weak alignment of COMET-src and the absence of correlation for GEMBA-DA-noref therefore indicate that these general-domain evaluation metrics do not fully reflect the clinical relevance and semantic precision valued by the human expert, highlighting a mismatch between general-domain evaluation metrics and domain-specific translation requirements.</Pgraph><SubHeadline>3.4 Ablation study: Assessing the concept extraction tool</SubHeadline><Pgraph>We manually evaluated a subset of 30 reports to assess the commercial entity recognition component. For each report, we annotated extracted concepts as true positives or false positives, and identified false negatives (missed concepts). From the 30 reports, the system extracted 419 concepts. Manual annotation showed 389 true positives and 30 false positives, yielding a precision of 92.8&#37;. We identified 161 false negatives, resulting in a recall of 70.7&#37; and an F1-score of 80.2&#37;. The false positives comprised mostly incorrect normalization mappings (e.g., &#8220;LAT&#8220; &#8594; &#8220;ORC3 protein, human&#8220; instead of &#8220;lateral&#8220;, &#8220;lead&#8220; &#8594; &#8220;Plumbum metallicum, homeopathic&#8220; instead of a medical device component, &#8220;MVR&#8220; &#8594; &#8220;Missing Value Reason&#8220; rather than &#8220;mitral valve replacement&#8220;, &#8220;Heart&#8220; &#8594; &#8220;HEART PROBLEM&#8220;).</Pgraph><Pgraph>The 161 false negatives comprised anatomical structures (&#34;cardiac silhouette&#8220;, &#8220;hilar contours&#8220;), clinical findings (&#34;pulmonary edema&#8220;, &#8220;atelectasis&#8220;, &#8220;pleural effusion&#8220;), medical devices (e.g., &#8220;dual chamber PPM&#8220;, &#8220;ETT placement&#8220;), and specific adjectives (e.g., &#8220;bronchovascular&#8220;, &#8220;costophrenic&#8220;). Many concepts that were missed in some reports were correctly identified in others, suggesting context-dependent recognition challenges rather than systematic failures, potentially aggravated by the rather low word count of radiology reports. See Appendix F (Attachment 1 <AttachmentLink attachmentNo="1" />) for a detailed list of false positives and false negatives.</Pgraph></TextBlock>
    <TextBlock name="4 Conclusion" linked="yes">
      <MainHeadline>4 Conclusion</MainHeadline><Pgraph>This study introduced Semantic Retrieval-Augmented Translation (S-RAT), a privacy-preserving approach that integrates structured clinical ontologies into machine translation workflows for clinical text. By leveraging UMLS and SNOMED CT concept mappings, S-RAT aims to improve terminology fidelity in radiology report translation without requiring cloud APIs or fine-tuning. We evaluated its feasibility, translation performance, and the reliability of automated evaluation metrics (COMET-src, GEMBA-DA-noref) against human judgment.</Pgraph><Pgraph>Overall, our findings were mixed. First, contrary to <Mark1>H1</Mark1>, the German edition of SNOMED CT did not provide sufficient concept coverage for radiological use cases, with only 32.64&#37; of extracted concepts successfully translated. While most distinct concept types were represented, frequent high-value terms such as <Mark2>Chest</Mark2> and <Mark2>Consolidation</Mark2> were systematically untranslated, which strongly limited overall coverage. These findings highlight the importance of domain-specific curation of a dictionary (see S-RAT-Dict and S-RAT-Curated) before ontology-based translation can be deployed in production environments.</Pgraph><Pgraph>The low coverage may partly reflect limitations of the German SNOMED CT edition rather than SNOMED CT itself. An edition with more complete target-language terminology could potentially yield different S-RAT results; however, we did not compare language editions directly. Future work should therefore assess whether terminology coverage is associated with translation performance.</Pgraph><Pgraph>Second, in relation to <Mark1>H2</Mark1>, retrieval-augmented prompting did not consistently outperform standard prompting. The lack of significant improvement might be driven by the prompting strategy itself. In several cases, manually curated or dictionary-based term injection helped preserve domain-specific phrasing but introduced occasional syntactic inconsistencies. Future work could explore adaptive prompt templates that dynamically adjust insertion granularity or confidence-weight concept injection based on retrieval certainty. Third, regarding <Mark1>H3</Mark1>, COMET-src correlated only weakly with expert adequacy ratings, while GEMBA-DA-noref showed no meaningful correlation. These findings indicate that general-domain reference-free metrics incompletely capture clinical translation quality. Although the weak yet statistically significant correlation suggests that COMET-src captures some degree of semantic alignment, it does not reflect the nuances of clinical correctness and factual precision that domain experts prioritize. This observation echoes similar findings in clinical summarization research, where general-domain metrics have shown poor alignment with human preferences for correctness and completeness <TextLink reference="27"></TextLink>. Developing or fine-tuning MT quality estimation models on biomedical or radiological corpora could thus be a promising direction.</Pgraph><Pgraph>From a methodological perspective, the S-RAT framework demonstrates the feasibility of integrating structured knowledge into small, locally deployed LLMs. While the current implementation did not yield significant gains, it represents an important step toward data-sovereign medical translation, where hospitals can maintain control over sensitive data while still benefiting from LLM-based translation. The lightweight retrieval and prompting mechanism is model-agnostic and could be integrated into future hospital-internal translation workflows.</Pgraph><Pgraph>Future work could address several extensions, including the enrichment of SNOMED CT with missing radiological terminology via semi-automated LLM translation of uncovered concepts, incorporating human feedback to iteratively refine prompts, and exploring self-assessment mechanisms that enable the model to flag uncertain translations. Beyond radiology, the approach could be generalized to other domains (e.g., pathology or cardiology) where high-precision, ontology-linked translations are critical for patient safety.</Pgraph><Pgraph>In conclusion, while S-RAT did not yet outperform standard prompting in quantitative metrics, it provides a conceptual and technical foundation for ontology-augmented, privacy-compliant medical translation. The study highlights both the potential and the current infrastructural limitations of structured retrieval for LLM-driven clinical applications. Our findings call for stronger multilingual standardization efforts within SNOMED CT and more domain-specific evaluation frameworks for MT applied in healthcare settings.</Pgraph></TextBlock>
    <TextBlock name="Notes" linked="yes">
      <MainHeadline>Notes</MainHeadline><SubHeadline>Competing interests</SubHeadline><Pgraph>The authors declare that they have no competing interests.</Pgraph><SubHeadline>Authors&#8217; ORCIDs</SubHeadline><Pgraph><UnorderedList><ListItem level="1">Daniel Reichenpfader: <Hyperlink href="https:&#47;&#47;orcid.org&#47;0000-0002-8052-3359">0000-0002-8052-3359</Hyperlink> </ListItem><ListItem level="1">Fabio Dennst&#228;dt: <Hyperlink href="https:&#47;&#47;orcid.org&#47;0000-0002-5374-8720">0000-0002-5374-8720</Hyperlink></ListItem></UnorderedList></Pgraph><SubHeadline>Author contributions</SubHeadline><Pgraph>DR and FD contributed to the conception and execution of the study, as well as data collection and interpretation. DR was responsible for study design, data analysis, and interpretation. DR drafted the manuscript, and FD critically revised it for important intellectual content. Both authors approved the final version of the manuscript and take responsibility for the scientific integrity of the work.</Pgraph></TextBlock>
    <References linked="yes">
      <Reference refNo="1">
        <RefAuthor>Kong M</RefAuthor>
        <RefAuthor>Fernandez A</RefAuthor>
        <RefAuthor>Bains J</RefAuthor>
        <RefAuthor>Milisavljevic A</RefAuthor>
        <RefAuthor>Brooks KC</RefAuthor>
        <RefAuthor>Shanmugam A</RefAuthor>
        <RefAuthor>Avilez L</RefAuthor>
        <RefAuthor>Li J</RefAuthor>
        <RefAuthor>Honcharov V</RefAuthor>
        <RefAuthor>Yang A</RefAuthor>
        <RefAuthor>Khoong EC</RefAuthor>
        <RefTitle>Evaluation of the accuracy and safety of machine translation of patient-specific discharge instructions: a comparative analysis</RefTitle>
        <RefYear>2026</RefYear>
        <RefJournal>BMJ Qual Saf</RefJournal>
        <RefPage>150-8</RefPage>
        <RefTotal>Kong M, Fernandez A, Bains J, Milisavljevic A, Brooks KC, Shanmugam A, Avilez L, Li J, Honcharov V, Yang A, Khoong EC. Evaluation of the accuracy and safety of machine translation of patient-specific discharge instructions: a comparative analysis. BMJ Qual Saf. 2026 Feb 19;35(3):150-8. DOI: 10.1136&#47;bmjqs-2024-018384</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1136&#47;bmjqs-2024-018384</RefLink>
      </Reference>
      <Reference refNo="2">
        <RefAuthor>Team Gemma</RefAuthor>
        <RefTitle>Gemma 3 Technical Report &#91;Preprint&#93;</RefTitle>
        <RefYear>2025</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefArticleNo>arXiv:2503.19786 &#91;cs&#93;</RefArticleNo>
        <RefTotal>Team Gemma. Gemma 3 Technical Report &#91;Preprint&#93;. arXiv. 2025: arXiv:2503.19786 &#91;cs&#93;. DOI: 10.48550&#47;arXiv.2503.19786</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.2503.19786</RefLink>
      </Reference>
      <Reference refNo="3">
        <RefAuthor>OpenAI</RefAuthor>
        <RefAuthor>Agarwal S</RefAuthor>
        <RefAuthor>Ahmad L</RefAuthor>
        <RefAuthor>Ai J</RefAuthor>
        <RefAuthor>Altman S</RefAuthor>
        <RefAuthor>Applebaum A</RefAuthor>
        <RefAuthor></RefAuthor>
        <RefTitle>gpt-oss-120b &#38; gpt-oss-20b Model Card &#91;Preprint&#93;</RefTitle>
        <RefYear>2025</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefArticleNo>arXiv:2508.10925 &#91;cs&#93;</RefArticleNo>
        <RefTotal>OpenAI, Agarwal S, Ahmad L, Ai J, Altman S, Applebaum A, et al. gpt-oss-120b &#38; gpt-oss-20b Model Card &#91;Preprint&#93;. arXiv. 2025: arXiv:2508.10925 &#91;cs&#93;. DOI: 10.48550&#47;arXiv.2508.10925</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.2508.10925</RefLink>
      </Reference>
      <Reference refNo="4">
        <RefAuthor>Apertus Team</RefAuthor>
        <RefTitle></RefTitle>
        <RefYear></RefYear>
        <RefBookTitle>Apertus: Democratizing open and compliant llms for global language environments</RefBookTitle>
        <RefPage></RefPage>
        <RefTotal>Apertus Team. Apertus: Democratizing open and compliant llms for global language environments. &#91;last accessed 2026 Aug 20&#93;. Available from: https:&#47;&#47;huggingface.co&#47;swiss-ai&#47;Apertus-70B-2509</RefTotal>
        <RefLink>https:&#47;&#47;huggingface.co&#47;swiss-ai&#47;Apertus-70B-2509</RefLink>
      </Reference>
      <Reference refNo="5">
        <RefAuthor>Bodenreider O</RefAuthor>
        <RefTitle>The Unified Medical Language System (UMLS): integrating biomedical terminology</RefTitle>
        <RefYear>2004</RefYear>
        <RefJournal>Nucleic Acids Res</RefJournal>
        <RefPage>D267-70</RefPage>
        <RefTotal>Bodenreider O. The Unified Medical Language System (UMLS): integrating biomedical terminology. Nucleic Acids Res. 2004 Jan;32(Database issue):D267-70. DOI: 10.1093&#47;nar&#47;gkh061</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1093&#47;nar&#47;gkh061</RefLink>
      </Reference>
      <Reference refNo="6">
        <RefAuthor>Meddeb A</RefAuthor>
        <RefAuthor>L&#252;ken S</RefAuthor>
        <RefAuthor>Busch F</RefAuthor>
        <RefAuthor>Adams L</RefAuthor>
        <RefAuthor>Ugga L</RefAuthor>
        <RefAuthor>Koltsakis E</RefAuthor>
        <RefAuthor>Tzortzakakis A</RefAuthor>
        <RefAuthor>Jelassi S</RefAuthor>
        <RefAuthor>Dkhil I</RefAuthor>
        <RefAuthor>Klontzas ME</RefAuthor>
        <RefAuthor>Triantafyllou M</RefAuthor>
        <RefAuthor>Kocak B</RefAuthor>
        <RefAuthor>Y&#252;zkan S</RefAuthor>
        <RefAuthor>Zhang L</RefAuthor>
        <RefAuthor>Hu B</RefAuthor>
        <RefAuthor>Andreychenko A</RefAuthor>
        <RefAuthor>Yurievich EA</RefAuthor>
        <RefAuthor>Logunova T</RefAuthor>
        <RefAuthor>Morakote W</RefAuthor>
        <RefAuthor>Angkurawaranon S</RefAuthor>
        <RefAuthor>Makowski MR</RefAuthor>
        <RefAuthor>Wattjes MP</RefAuthor>
        <RefAuthor>Cuocolo R</RefAuthor>
        <RefAuthor>Bressem K</RefAuthor>
        <RefTitle>Large Language Model Ability to Translate CT and MRI Free-Text Radiology Reports Into Multiple Languages</RefTitle>
        <RefYear>2024</RefYear>
        <RefJournal>Radiology</RefJournal>
        <RefPage>e241736</RefPage>
        <RefTotal>Meddeb A, L&#252;ken S, Busch F, Adams L, Ugga L, Koltsakis E, Tzortzakakis A, Jelassi S, Dkhil I, Klontzas ME, Triantafyllou M, Kocak B, Y&#252;zkan S, Zhang L, Hu B, Andreychenko A, Yurievich EA, Logunova T, Morakote W, Angkurawaranon S, Makowski MR, Wattjes MP, Cuocolo R, Bressem K. Large Language Model Ability to Translate CT and MRI Free-Text Radiology Reports Into Multiple Languages. Radiology. 2024 Dec;313(3):e241736. 
DOI: 10.1148&#47;radiol.241736</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1148&#47;radiol.241736</RefLink>
      </Reference>
      <Reference refNo="7">
        <RefAuthor>Merx R</RefAuthor>
        <RefAuthor>Suominen H</RefAuthor>
        <RefAuthor>Cohn T</RefAuthor>
        <RefAuthor>Vylomova E</RefAuthor>
        <RefTitle>OpenWHO: A Document-Level Parallel Corpus for Health Translation in Low-Resource Languages &#91;Preprint&#93;</RefTitle>
        <RefYear>2025</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefArticleNo>arXiv:2508.16048 &#91;cs&#93;</RefArticleNo>
        <RefTotal>Merx R, Suominen H, Cohn T, Vylomova E. OpenWHO: A Document-Level Parallel Corpus for Health Translation in Low-Resource Languages &#91;Preprint&#93;. arXiv. 2025: arXiv:2508.16048 &#91;cs&#93;. DOI: 10.48550&#47;arXiv.2508.16048</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.2508.16048</RefLink>
      </Reference>
      <Reference refNo="8">
        <RefAuthor>Garc&#237;a-Ferrero I</RefAuthor>
        <RefAuthor>Agerri R</RefAuthor>
        <RefAuthor>Salazar AA</RefAuthor>
        <RefAuthor>Cabrio E</RefAuthor>
        <RefAuthor>Iglesia Idl</RefAuthor>
        <RefAuthor>Lavelli A</RefAuthor>
        <RefAuthor></RefAuthor>
        <RefTitle>Medical mT5: An Open-Source Multilingual Text-to-Text LLM for The Medical Domain &#91;Preprint&#93;</RefTitle>
        <RefYear>2024</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefArticleNo>arXiv:2404.07613 &#91;cs&#93;</RefArticleNo>
        <RefTotal>Garc&#237;a-Ferrero I, Agerri R, Salazar AA, Cabrio E, Iglesia Idl, Lavelli A, et al. Medical mT5: An Open-Source Multilingual Text-to-Text LLM for The Medical Domain &#91;Preprint&#93;. arXiv. 2024: arXiv:2404.07613 &#91;cs&#93;. DOI: 10.48550&#47;arXiv.2404.07613</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.2404.07613</RefLink>
      </Reference>
      <Reference refNo="9">
        <RefAuthor>Vo N</RefAuthor>
        <RefAuthor>Nguyen DQ</RefAuthor>
        <RefAuthor>Le DD</RefAuthor>
        <RefAuthor>Piccardi M</RefAuthor>
        <RefAuthor>Buntine W</RefAuthor>
        <RefTitle>Improving Vietnamese-English Medical Machine Translation &#91;Preprint&#93;</RefTitle>
        <RefYear>2024</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefArticleNo>arXiv:2403.19161 &#91;cs&#93;</RefArticleNo>
        <RefTotal>Vo N, Nguyen DQ, Le DD, Piccardi M, Buntine W. Improving Vietnamese-English Medical Machine Translation &#91;Preprint&#93;. arXiv. 2024: arXiv:2403.19161 &#91;cs&#93;. 
DOI: 10.48550&#47;arXiv.2403.19161</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.2403.19161</RefLink>
      </Reference>
      <Reference refNo="10">
        <RefAuthor>Rios M</RefAuthor>
        <RefTitle>Instruction-tuned Large Language Models for Machine Translation in the Medical Domain &#91;Preprint&#93;</RefTitle>
        <RefYear>2024</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefArticleNo>arXiv:2408.16440v1 &#91;cs&#93;</RefArticleNo>
        <RefTotal>Rios M. Instruction-tuned Large Language Models for Machine Translation in the Medical Domain &#91;Preprint&#93;. arXiv. 2024: arXiv:2408.16440v1 &#91;cs&#93;. DOI: 10.48550&#47;arXiv.2408.16440</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.2408.16440</RefLink>
      </Reference>
      <Reference refNo="11">
        <RefAuthor>Lewis P</RefAuthor>
        <RefAuthor>Perez E</RefAuthor>
        <RefAuthor>Piktus A</RefAuthor>
        <RefAuthor>Petroni F</RefAuthor>
        <RefAuthor>Karpukhin V</RefAuthor>
        <RefAuthor>Goyal N</RefAuthor>
        <RefAuthor></RefAuthor>
        <RefTitle>Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks &#91;Preprint&#93;</RefTitle>
        <RefYear>2021</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefArticleNo>arXiv:2005.11401 &#91;cs&#93;</RefArticleNo>
        <RefTotal>Lewis P, Perez E, Piktus A, Petroni F, Karpukhin V, Goyal N, et al. Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks &#91;Preprint&#93;. arXiv. 2021: arXiv:2005.11401 &#91;cs&#93;. 
DOI: 10.48550&#47;arXiv.2005.11401</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.2005.11401</RefLink>
      </Reference>
      <Reference refNo="12">
        <RefAuthor>Wang J</RefAuthor>
        <RefAuthor>Meng F</RefAuthor>
        <RefAuthor>Zhang Y</RefAuthor>
        <RefAuthor>Zhou J</RefAuthor>
        <RefTitle>Retrieval-Augmented Machine Translation with Unstructured Knowledge &#91;Preprint&#93;</RefTitle>
        <RefYear>2024</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefArticleNo>arXiv:2412.04342v1 &#91;cs&#93;</RefArticleNo>
        <RefTotal>Wang J, Meng F, Zhang Y, Zhou J. Retrieval-Augmented Machine Translation with Unstructured Knowledge &#91;Preprint&#93;. arXiv. 2024: arXiv:2412.04342v1 &#91;cs&#93;. DOI: 10.48550&#47;arXiv.2412.04342</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.2412.04342</RefLink>
      </Reference>
      <Reference refNo="13">
        <RefAuthor>Microsoft Azure</RefAuthor>
        <RefTitle></RefTitle>
        <RefYear></RefYear>
        <RefBookTitle>What is the Text Analytics for health in Azure AI Language&#63;</RefBookTitle>
        <RefPage></RefPage>
        <RefTotal>Microsoft Azure. What is the Text Analytics for health in Azure AI Language&#63; &#91;last accessed 2026 Aug 20&#93;. Available from: https:&#47;&#47;learn.microsoft.com&#47;en-us&#47;azure&#47;ai-services&#47;language-service&#47;text-analytics-for-health&#47;overview</RefTotal>
        <RefLink>https:&#47;&#47;learn.microsoft.com&#47;en-us&#47;azure&#47;ai-services&#47;language-service&#47;text-analytics-for-health&#47;overview</RefLink>
      </Reference>
      <Reference refNo="14">
        <RefAuthor>NIH National Library of Medicine</RefAuthor>
        <RefTitle></RefTitle>
        <RefYear></RefYear>
        <RefBookTitle>UMLS API Home</RefBookTitle>
        <RefPage></RefPage>
        <RefTotal>NIH National Library of Medicine. UMLS API Home. &#91;last accessed 2026 Aug 20&#93;. Available from: https:&#47;&#47;documentation.uts.nlm.nih.gov&#47;rest&#47;home.html</RefTotal>
        <RefLink>https:&#47;&#47;documentation.uts.nlm.nih.gov&#47;rest&#47;home.html</RefLink>
      </Reference>
      <Reference refNo="15">
        <RefAuthor>SNOMED International</RefAuthor>
        <RefTitle></RefTitle>
        <RefYear></RefYear>
        <RefBookTitle>IHTSDO&#47;snowstorm</RefBookTitle>
        <RefPage></RefPage>
        <RefTotal>SNOMED International. IHTSDO&#47;snowstorm. Github; &#91;last accessed 2026 Aug 20&#93;. Available from: https:&#47;&#47;github.com&#47;IHTSDO&#47;snowstorm</RefTotal>
        <RefLink>https:&#47;&#47;github.com&#47;IHTSDO&#47;snowstorm</RefLink>
      </Reference>
      <Reference refNo="16">
        <RefAuthor>Johnson AEW</RefAuthor>
        <RefAuthor>Pollard TJ</RefAuthor>
        <RefAuthor>Berkowitz SJ</RefAuthor>
        <RefAuthor>Greenbaum NR</RefAuthor>
        <RefAuthor>Lungren MP</RefAuthor>
        <RefAuthor>Deng CY</RefAuthor>
        <RefAuthor>Mark RG</RefAuthor>
        <RefAuthor>Horng S</RefAuthor>
        <RefTitle>MIMIC-CXR, a de-identified publicly available database of chest radiographs with free-text reports</RefTitle>
        <RefYear>2019</RefYear>
        <RefJournal>Sci Data</RefJournal>
        <RefPage>317</RefPage>
        <RefTotal>Johnson AEW, Pollard TJ, Berkowitz SJ, Greenbaum NR, Lungren MP, Deng CY, Mark RG, Horng S. MIMIC-CXR, a de-identified publicly available database of chest radiographs with free-text reports. Sci Data. 2019 Dec;6(1):317. DOI: 10.1038&#47;s41597-019-0322-0</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1038&#47;s41597-019-0322-0</RefLink>
      </Reference>
      <Reference refNo="17">
        <RefAuthor>Christophe C</RefAuthor>
        <RefAuthor>Kanithi PK</RefAuthor>
        <RefAuthor>Raha T</RefAuthor>
        <RefAuthor>Khan S</RefAuthor>
        <RefAuthor>Pimentel MA</RefAuthor>
        <RefTitle>Med42-v2: A Suite of Clinical LLMs &#91;Preprint&#93;</RefTitle>
        <RefYear>2024</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefArticleNo>arXiv:2408.06142 &#91;cs&#93;</RefArticleNo>
        <RefTotal>Christophe C, Kanithi PK, Raha T, Khan S, Pimentel MA. Med42-v2: A Suite of Clinical LLMs &#91;Preprint&#93;. arXiv. 2024: arXiv:2408.06142 &#91;cs&#93;. DOI: 10.48550&#47;arXiv.2408.06142</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.2408.06142</RefLink>
      </Reference>
      <Reference refNo="18">
        <RefAuthor>DeepSeek-AI</RefAuthor>
        <RefAuthor>Guo D</RefAuthor>
        <RefAuthor>Yang D</RefAuthor>
        <RefAuthor>Zhang H</RefAuthor>
        <RefAuthor>Song J</RefAuthor>
        <RefAuthor>Zhang R</RefAuthor>
        <RefAuthor></RefAuthor>
        <RefTitle>DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning &#91;Preprint&#93;</RefTitle>
        <RefYear>2025</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefArticleNo>arXiv:2501.12948 &#91;cs&#93;</RefArticleNo>
        <RefTotal>DeepSeek-AI, Guo D, Yang D, Zhang H, Song J, Zhang R, et al. DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning &#91;Preprint&#93;. arXiv. 2025: arXiv:2501.12948 &#91;cs&#93;. DOI: 10.48550&#47;arXiv.2501.12948</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.2501.12948</RefLink>
      </Reference>
      <Reference refNo="19">
        <RefAuthor>Google DeepMind</RefAuthor>
        <RefTitle></RefTitle>
        <RefYear>2025</RefYear>
        <RefBookTitle>Gemini 3 Pro Model Card</RefBookTitle>
        <RefPage></RefPage>
        <RefTotal>Google DeepMind. Gemini 3 Pro Model Card. 2025 Nov &#91;last updated 2026 May&#93;. Available from: https:&#47;&#47;storage.googleapis.com&#47;deepmind-media&#47;Model-Cards&#47;Gemini-3-Pro-Model-Card.pdf</RefTotal>
        <RefLink>https:&#47;&#47;storage.googleapis.com&#47;deepmind-media&#47;Model-Cards&#47;Gemini-3-Pro-Model-Card.pdf</RefLink>
      </Reference>
      <Reference refNo="20">
        <RefAuthor>Chen CL</RefAuthor>
        <RefAuthor>Dong Y</RefAuthor>
        <RefAuthor>Castillo-Zambrano C</RefAuthor>
        <RefAuthor>Bencheqroun H</RefAuthor>
        <RefAuthor>Barwise A</RefAuthor>
        <RefAuthor>Hoffman A</RefAuthor>
        <RefAuthor>Nalaie K</RefAuthor>
        <RefAuthor>Qiu Y</RefAuthor>
        <RefAuthor>Boulekbache O</RefAuthor>
        <RefAuthor>Niven AS</RefAuthor>
        <RefTitle>A systematic multimodal assessment of AI machine translation tools for enhancing access to critical care education internationally</RefTitle>
        <RefYear>2025</RefYear>
        <RefJournal>BMC Med Educ</RefJournal>
        <RefPage>1022</RefPage>
        <RefTotal>Chen CL, Dong Y, Castillo-Zambrano C, Bencheqroun H, Barwise A, Hoffman A, Nalaie K, Qiu Y, Boulekbache O, Niven AS. A systematic multimodal assessment of AI machine translation tools for enhancing access to critical care education internationally. BMC Med Educ. 2025 Jul;25(1):1022. 
DOI: 10.1186&#47;s12909-025-07452-9</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1186&#47;s12909-025-07452-9</RefLink>
      </Reference>
      <Reference refNo="21">
        <RefAuthor>Brown T</RefAuthor>
        <RefAuthor>Mann B</RefAuthor>
        <RefAuthor>Ryder N</RefAuthor>
        <RefAuthor>Subbiah M</RefAuthor>
        <RefAuthor>Kaplan JD</RefAuthor>
        <RefAuthor>Dhariwal P</RefAuthor>
        <RefAuthor></RefAuthor>
        <RefTitle>Language Models are Few-Shot Learners</RefTitle>
        <RefYear>2020</RefYear>
        <RefBookTitle>Advances in Neural Information Processing Systems 33. NeurIPS 2020.</RefBookTitle>
        <RefPage>1877-901</RefPage>
        <RefTotal>Brown T, Mann B, Ryder N, Subbiah M, Kaplan JD, Dhariwal P, et al. Language Models are Few-Shot Learners. In: Larochelle H, Ranzato M, Hadsell R, Balcan MF, Lin H, editors. Advances in Neural Information Processing Systems 33. NeurIPS 2020. Curran Associates, Inc.; 2020 &#91;last accessed 2026 Aug 20&#93;. p. 1877-901. Available from: https:&#47;&#47;proceedings.neurips.cc&#47;paper&#95;files&#47;paper&#47;2020&#47;hash&#47;1457c0d6bfcb4967418bfb8ac142f64a-Abstract.html</RefTotal>
        <RefLink>https:&#47;&#47;proceedings.neurips.cc&#47;paper&#95;files&#47;paper&#47;2020&#47;hash&#47;1457c0d6bfcb4967418bfb8ac142f64a-Abstract.html</RefLink>
      </Reference>
      <Reference refNo="23">
        <RefAuthor>Papineni K</RefAuthor>
        <RefAuthor>Roukos S</RefAuthor>
        <RefAuthor>Ward T</RefAuthor>
        <RefAuthor>Zhu WJ</RefAuthor>
        <RefTitle>Bleu: a Method for Automatic Evaluation of Machine Translation</RefTitle>
        <RefYear>2002</RefYear>
        <RefBookTitle>Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics</RefBookTitle>
        <RefPage>311-8</RefPage>
        <RefTotal>Papineni K, Roukos S, Ward T, Zhu WJ. Bleu: a Method for Automatic Evaluation of Machine Translation. In: Isabelle P, Charniak E, Lin D, editors. Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics. Association for Computational Linguistics; 2002. p. 311-8. 
DOI: 10.3115&#47;1073083.1073135</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.3115&#47;1073083.1073135</RefLink>
      </Reference>
      <Reference refNo="24">
        <RefAuthor>Lin CY</RefAuthor>
        <RefTitle>ROUGE: A Package for Automatic Evaluation of Summaries</RefTitle>
        <RefYear>2004</RefYear>
        <RefBookTitle>Text Summarization Branches Out</RefBookTitle>
        <RefPage>74-81</RefPage>
        <RefTotal>Lin CY. ROUGE: A Package for Automatic Evaluation of Summaries. In: Text Summarization Branches Out. Association for Computational Linguistics; 2004 &#91;last accessed 2026 Aug 20&#93;. p. 74-81. Available from: https:&#47;&#47;aclanthology.org&#47;W04-1013&#47;</RefTotal>
        <RefLink>https:&#47;&#47;aclanthology.org&#47;W04-1013&#47;</RefLink>
      </Reference>
      <Reference refNo="25">
        <RefAuthor>Banerjee S</RefAuthor>
        <RefAuthor>Lavie A</RefAuthor>
        <RefTitle>METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments</RefTitle>
        <RefYear>2005</RefYear>
        <RefBookTitle>Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and&#47;or Summarization</RefBookTitle>
        <RefPage>65-72</RefPage>
        <RefTotal>Banerjee S, Lavie A. METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments. In: Goldstein J, Lavie A, Lin CY, Voss C, editors. Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and&#47;or Summarization. Association for Computational Linguistics; 2005 &#91;last accessed 2026 Aug 20&#93;. p. 65-72. Available from: https:&#47;&#47;aclanthology.org&#47;W05-0909&#47;</RefTotal>
        <RefLink>https:&#47;&#47;aclanthology.org&#47;W05-0909&#47;</RefLink>
      </Reference>
      <Reference refNo="26">
        <RefAuthor>Zhang T</RefAuthor>
        <RefAuthor>Kishore V</RefAuthor>
        <RefAuthor>Wu F</RefAuthor>
        <RefAuthor>Weinberger KQ</RefAuthor>
        <RefAuthor>Artzi Y</RefAuthor>
        <RefTitle>BERTScore: Evaluating Text Generation with BERT &#91;Preprint&#93;</RefTitle>
        <RefYear>2020</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefArticleNo>arXiv:1904.09675 &#91;cs&#93;</RefArticleNo>
        <RefTotal>Zhang T, Kishore V, Wu F, Weinberger KQ, Artzi Y. BERTScore: Evaluating Text Generation with BERT &#91;Preprint&#93;. arXiv. 2020: arXiv:1904.09675 &#91;cs&#93;. DOI: 10.48550&#47;arXiv.1904.09675</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.1904.09675</RefLink>
      </Reference>
      <Reference refNo="27">
        <RefAuthor>Van Veen D</RefAuthor>
        <RefAuthor>Van Uden C</RefAuthor>
        <RefAuthor>Blankemeier L</RefAuthor>
        <RefAuthor>Delbrouck JB</RefAuthor>
        <RefAuthor>Aali A</RefAuthor>
        <RefAuthor>Bluethgen C</RefAuthor>
        <RefAuthor>Pareek A</RefAuthor>
        <RefAuthor>Polacin M</RefAuthor>
        <RefAuthor>Reis EP</RefAuthor>
        <RefAuthor>Seehofnerov&#225; A</RefAuthor>
        <RefAuthor>Rohatgi N</RefAuthor>
        <RefAuthor>Hosamani P</RefAuthor>
        <RefAuthor>Collins W</RefAuthor>
        <RefAuthor>Ahuja N</RefAuthor>
        <RefAuthor>Langlotz CP</RefAuthor>
        <RefAuthor>Hom J</RefAuthor>
        <RefAuthor>Gatidis S</RefAuthor>
        <RefAuthor>Pauly J</RefAuthor>
        <RefAuthor>Chaudhari AS</RefAuthor>
        <RefTitle>Adapted large language models can outperform medical experts in clinical text summarization</RefTitle>
        <RefYear>2024</RefYear>
        <RefJournal>Nat Med</RefJournal>
        <RefPage>1134-42</RefPage>
        <RefTotal>Van Veen D, Van Uden C, Blankemeier L, Delbrouck JB, Aali A, Bluethgen C, Pareek A, Polacin M, Reis EP, Seehofnerov&#225; A, Rohatgi N, Hosamani P, Collins W, Ahuja N, Langlotz CP, Hom J, Gatidis S, Pauly J, Chaudhari AS. Adapted large language models can outperform medical experts in clinical text summarization. Nat Med. 2024 Apr;30(4):1134-42. DOI: 10.1038&#47;s41591-024-02855-5</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1038&#47;s41591-024-02855-5</RefLink>
      </Reference>
      <Reference refNo="28">
        <RefAuthor>Zhao H</RefAuthor>
        <RefAuthor>Liu Y</RefAuthor>
        <RefAuthor>Tao S</RefAuthor>
        <RefAuthor>Meng W</RefAuthor>
        <RefAuthor>Chen Y</RefAuthor>
        <RefAuthor>Geng X</RefAuthor>
        <RefAuthor></RefAuthor>
        <RefTitle>From Handcrafted Features to LLMs: A Brief Survey for Machine Translation Quality Estimation</RefTitle>
        <RefYear>2024</RefYear>
        <RefBookTitle>2024 International Joint Conference on Neural Networks (IJCNN); 2024 Jun 30 - Jul 05; Yokohama, Japan</RefBookTitle>
        <RefPage></RefPage>
        <RefTotal>Zhao H, Liu Y, Tao S, Meng W, Chen Y, Geng X, et al. From Handcrafted Features to LLMs: A Brief Survey for Machine Translation Quality Estimation. In: 2024 International Joint Conference on Neural Networks (IJCNN); 2024 Jun 30 - Jul 05; Yokohama, Japan. IEEE; 2024. 
DOI: 10.1109&#47;IJCNN60899.2024.10650457</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1109&#47;IJCNN60899.2024.10650457</RefLink>
      </Reference>
      <Reference refNo="29">
        <RefAuthor>Lo Ck</RefAuthor>
        <RefTitle>YiSi - a Unified Semantic MT Quality Evaluation and Estimation Metric for Languages with Different Levels of Available Resources</RefTitle>
        <RefYear>2019</RefYear>
        <RefBookTitle>Proceedings of the Fourth Conference on Machine Translation (Volume 2: Shared Task Papers, Day 1)</RefBookTitle>
        <RefPage>507-13</RefPage>
        <RefTotal>Lo Ck. YiSi - a Unified Semantic MT Quality Evaluation and Estimation Metric for Languages with Different Levels of Available Resources. In: Bojar O, Chatterjee R, Federmann C, Fishel M, Graham Y, Haddow B, et al, editors. Proceedings of the Fourth Conference on Machine Translation (Volume 2: Shared Task Papers, Day 1). Association for Computational Linguistics; 2019. p. 507-13. DOI: 10.18653&#47;v1&#47;W19-5358</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.18653&#47;v1&#47;W19-5358</RefLink>
      </Reference>
      <Reference refNo="30">
        <RefAuthor>Thompson B</RefAuthor>
        <RefAuthor>Post M</RefAuthor>
        <RefTitle>Automatic Machine Translation Evaluation in Many Languages via Zero-Shot Paraphrasing</RefTitle>
        <RefYear>2020</RefYear>
        <RefBookTitle>Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP)</RefBookTitle>
        <RefPage>90-121</RefPage>
        <RefTotal>Thompson B, Post M. Automatic Machine Translation Evaluation in Many Languages via Zero-Shot Paraphrasing. In: Webber B, Cohn T, He Y, Liu Y, editors. Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP). Association for Computational Linguistics; 2020. p. 90-121. 
DOI: 10.18653&#47;v1&#47;2020.emnlp-main.8</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.18653&#47;v1&#47;2020.emnlp-main.8</RefLink>
      </Reference>
      <Reference refNo="31">
        <RefAuthor>Song Y</RefAuthor>
        <RefAuthor>Zhao J</RefAuthor>
        <RefAuthor>Specia L</RefAuthor>
        <RefTitle>SentSim: Crosslingual Semantic Evaluation of Machine Translation</RefTitle>
        <RefYear>2021</RefYear>
        <RefBookTitle>Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies</RefBookTitle>
        <RefPage>3143-56</RefPage>
        <RefTotal>Song Y, Zhao J, Specia L. SentSim: Crosslingual Semantic Evaluation of Machine Translation. In: Toutanova K, Rumshisky A, Zettlemoyer L, Hakkani-Tur D, Beltagy I, Bethard S, et al, editors. Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. Association for Computational Linguistics; 2021. p. 3143-56. 
DOI: 10.18653&#47;v1&#47;2021.naacl-main.252</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.18653&#47;v1&#47;2021.naacl-main.252</RefLink>
      </Reference>
      <Reference refNo="32">
        <RefAuthor>Wu H</RefAuthor>
        <RefAuthor>Han W</RefAuthor>
        <RefAuthor>Di H</RefAuthor>
        <RefAuthor>Chen Y</RefAuthor>
        <RefAuthor>Xu J</RefAuthor>
        <RefTitle>A Holistic Approach to Reference-Free Evaluation of Machine Translation</RefTitle>
        <RefYear>2023</RefYear>
        <RefBookTitle>Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers)</RefBookTitle>
        <RefPage>623-36</RefPage>
        <RefTotal>Wu H, Han W, Di H, Chen Y, Xu J. A Holistic Approach to Reference-Free Evaluation of Machine Translation. In: Rogers A, Boyd-Graber J, Okazaki N, editors. Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers). Association for Computational Linguistics; 2023. p. 623-36. DOI: 10.18653&#47;v1&#47;2023.acl-short.55</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.18653&#47;v1&#47;2023.acl-short.55</RefLink>
      </Reference>
      <Reference refNo="33">
        <RefAuthor>Moosa IM</RefAuthor>
        <RefAuthor>Zhang R</RefAuthor>
        <RefAuthor>Yin W</RefAuthor>
        <RefTitle>MT-Ranker: Reference-free machine translation evaluation by inter-system ranking &#91;Preprint&#93;</RefTitle>
        <RefYear>2024</RefYear>
        <RefJournal>arXiv</RefJournal>
        <RefArticleNo>arXiv:2401.17099 &#91;cs&#93;</RefArticleNo>
        <RefTotal>Moosa IM, Zhang R, Yin W. MT-Ranker: Reference-free machine translation evaluation by inter-system ranking &#91;Preprint&#93;. arXiv. 2024: arXiv:2401.17099 &#91;cs&#93;. DOI: 10.48550&#47;arXiv.2401.17099</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.48550&#47;arXiv.2401.17099</RefLink>
      </Reference>
      <Reference refNo="34">
        <RefAuthor>Liu Y</RefAuthor>
        <RefAuthor>Iter D</RefAuthor>
        <RefAuthor>Xu Y</RefAuthor>
        <RefAuthor>Wang S</RefAuthor>
        <RefAuthor>Xu R</RefAuthor>
        <RefAuthor>Zhu C</RefAuthor>
        <RefTitle>G-Eval: NLG Evaluation using Gpt-4 with Better Human Alignment</RefTitle>
        <RefYear>2023</RefYear>
        <RefBookTitle>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing</RefBookTitle>
        <RefPage>2511-22</RefPage>
        <RefTotal>Liu Y, Iter D, Xu Y, Wang S, Xu R, Zhu C. G-Eval: NLG Evaluation using Gpt-4 with Better Human Alignment. In: Bouamor H, Pino J, Bali K, editors. Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics; 2023. p. 2511-22. 
DOI: 10.18653&#47;v1&#47;2023.emnlp-main.153</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.18653&#47;v1&#47;2023.emnlp-main.153</RefLink>
      </Reference>
      <Reference refNo="35">
        <RefAuthor>Zheng L</RefAuthor>
        <RefAuthor>Chiang WL</RefAuthor>
        <RefAuthor>Sheng Y</RefAuthor>
        <RefAuthor>Zhuang S</RefAuthor>
        <RefAuthor>Wu Z</RefAuthor>
        <RefAuthor>Zhuang Y</RefAuthor>
        <RefAuthor></RefAuthor>
        <RefTitle>Judging LLM-as-a-judge with MT-bench and Chatbot Arena</RefTitle>
        <RefYear>2023</RefYear>
        <RefBookTitle>Proceedings of the 37th International Conference on Neural Information Processing Systems. NIPS &#8217;23; 2023 Dec 10-16; New Orleans, Louisiana, USA</RefBookTitle>
        <RefPage>46595-623</RefPage>
        <RefTotal>Zheng L, Chiang WL, Sheng Y, Zhuang S, Wu Z, Zhuang Y, et al. Judging LLM-as-a-judge with MT-bench and Chatbot Arena. In: Proceedings of the 37th International Conference on Neural Information Processing Systems. NIPS &#8217;23; 2023 Dec 10-16; New Orleans, Louisiana, USA. Red Hook, NY, USA: Curran Associates Inc.; 2023. p. 46595-623. DOI: 10.52202&#47;075280-2020</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.52202&#47;075280-2020</RefLink>
      </Reference>
      <Reference refNo="36">
        <RefAuthor>Wang P</RefAuthor>
        <RefAuthor>Li L</RefAuthor>
        <RefAuthor>Chen L</RefAuthor>
        <RefAuthor>Cai Z</RefAuthor>
        <RefAuthor>Zhu D</RefAuthor>
        <RefAuthor>Lin B</RefAuthor>
        <RefAuthor></RefAuthor>
        <RefTitle>Large Language Models are not Fair Evaluators</RefTitle>
        <RefYear>2024</RefYear>
        <RefBookTitle>Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</RefBookTitle>
        <RefPage>9440-50</RefPage>
        <RefTotal>Wang P, Li L, Chen L, Cai Z, Zhu D, Lin B, et al. Large Language Models are not Fair Evaluators. In: Ku LW, Martins A, Srikumar V, editors. Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). Association for Computational Linguistics; 2024. 
p. 9440-50. DOI: 10.18653&#47;v1&#47;2024.acl-long.511</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.18653&#47;v1&#47;2024.acl-long.511</RefLink>
      </Reference>
      <Reference refNo="37">
        <RefAuthor>Kocmi T</RefAuthor>
        <RefAuthor>Federmann C</RefAuthor>
        <RefAuthor>Grundkiewicz R</RefAuthor>
        <RefAuthor>Junczys-Dowmunt M</RefAuthor>
        <RefAuthor>Matsushita H</RefAuthor>
        <RefAuthor>Menezes A</RefAuthor>
        <RefTitle>To Ship or Not to Ship: An Extensive Evaluation of Automatic Metrics for Machine Translation</RefTitle>
        <RefYear>2021</RefYear>
        <RefBookTitle>Proceedings of the Sixth Conference on Machine Translation</RefBookTitle>
        <RefPage>478-94</RefPage>
        <RefTotal>Kocmi T, Federmann C, Grundkiewicz R, Junczys-Dowmunt M, Matsushita H, Menezes A. To Ship or Not to Ship: An Extensive Evaluation of Automatic Metrics for Machine Translation. In: Barrault L, Bojar O, Bougares F, Chatterjee R, Costajussa MR, Federmann C, et al, editors. Proceedings of the Sixth Conference on Machine Translation. Association for Computational Linguistics; 2021 &#91;last accessed 2026 Aug 20&#93;. p. 478-94. Available from: https:&#47;&#47;aclanthology.org&#47;2021.wmt-1.57</RefTotal>
        <RefLink>https:&#47;&#47;aclanthology.org&#47;2021.wmt-1.57</RefLink>
      </Reference>
      <Reference refNo="38">
        <RefAuthor>Rei R</RefAuthor>
        <RefAuthor>Guerreiro NM</RefAuthor>
        <RefAuthor>Pombal J</RefAuthor>
        <RefAuthor>van Stigt D</RefAuthor>
        <RefAuthor>Treviso M</RefAuthor>
        <RefAuthor>Coheur L</RefAuthor>
        <RefAuthor></RefAuthor>
        <RefTitle>Scaling up CometKiwi: Unbabel-IST 2023 Submission for the Quality Estimation Shared Task</RefTitle>
        <RefYear>2023</RefYear>
        <RefBookTitle>Proceedings of the Eighth Conference on Machine Translation</RefBookTitle>
        <RefPage>841-8</RefPage>
        <RefTotal>Rei R, Guerreiro NM, Pombal J, van Stigt D, Treviso M, Coheur L, et al. Scaling up CometKiwi: Unbabel-IST 2023 Submission for the Quality Estimation Shared Task. In: Koehn P, Haddow B, Kocmi T, Monz C, editors. Proceedings of the Eighth Conference on Machine Translation. Association for Computational Linguistics; 2023. p. 841-8. DOI: 10.18653&#47;v1&#47;2023.wmt-1.73</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.18653&#47;v1&#47;2023.wmt-1.73</RefLink>
      </Reference>
      <Reference refNo="39">
        <RefAuthor>Kocmi T</RefAuthor>
        <RefAuthor>Avramidis E</RefAuthor>
        <RefAuthor>Bawden R</RefAuthor>
        <RefAuthor>Bojar O</RefAuthor>
        <RefAuthor>Dvorkovich A</RefAuthor>
        <RefAuthor>Federmann C</RefAuthor>
        <RefAuthor></RefAuthor>
        <RefTitle>Findings of the WMT24 General Machine Translation Shared Task: The LLM Era Is Here but MT Is Not Solved Yet</RefTitle>
        <RefYear>2024</RefYear>
        <RefBookTitle>Proceedings of the Ninth Conference on Machine Translation</RefBookTitle>
        <RefPage>1-46</RefPage>
        <RefTotal>Kocmi T, Avramidis E, Bawden R, Bojar O, Dvorkovich A, Federmann C, et al. Findings of the WMT24 General Machine Translation Shared Task: The LLM Era Is Here but MT Is Not Solved Yet. In: Haddow B, Kocmi T, Koehn P, Monz C, editors. Proceedings of the Ninth Conference on Machine Translation. Association for Computational Linguistics; 2024. p. 1-46. 
DOI: DOI: 10.18653&#47;v1&#47;2024.wmt-1.1</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.18653&#47;v1&#47;2024.wmt-1.1</RefLink>
      </Reference>
      <Reference refNo="40">
        <RefAuthor>Unbabel</RefAuthor>
        <RefTitle></RefTitle>
        <RefYear></RefYear>
        <RefBookTitle>Unbabel&#47;COMET</RefBookTitle>
        <RefPage></RefPage>
        <RefTotal>Unbabel. Unbabel&#47;COMET. Github; &#91;last accessed 2026 Aug 20&#93;. Available from: https:&#47;&#47;github.com&#47;Unbabel&#47;COMET</RefTotal>
        <RefLink>https:&#47;&#47;github.com&#47;Unbabel&#47;COMET</RefLink>
      </Reference>
      <Reference refNo="41">
        <RefAuthor>Kocmi T</RefAuthor>
        <RefAuthor>Federmann C</RefAuthor>
        <RefTitle>Large Language Models Are State-of-the-Art Evaluators of Translation Quality</RefTitle>
        <RefYear>2023</RefYear>
        <RefBookTitle>Proceedings of the 24th Annual Conference of the European Association for Machine Translation</RefBookTitle>
        <RefPage>193-203</RefPage>
        <RefTotal>Kocmi T, Federmann C. Large Language Models Are State-of-the-Art Evaluators of Translation Quality. In: Nurminen M, Brenner J, Koponen M, Latomaa S, Mikhailov M, Schierl F, et al, editors. Proceedings of the 24th Annual Conference of the European Association for Machine Translation. European Association for Machine Translation; 2023 &#91;last accessed 2026 Aug 20&#93;. p. 193-203. Available from: https:&#47;&#47;aclanthology.org&#47;2023.eamt-1.19&#47;</RefTotal>
        <RefLink>https:&#47;&#47;aclanthology.org&#47;2023.eamt-1.19&#47;</RefLink>
      </Reference>
      <Reference refNo="42">
        <RefAuthor>Lu Q</RefAuthor>
        <RefAuthor>Qiu B</RefAuthor>
        <RefAuthor>Ding L</RefAuthor>
        <RefAuthor>Zhang K</RefAuthor>
        <RefAuthor>Kocmi T</RefAuthor>
        <RefAuthor>Tao D</RefAuthor>
        <RefTitle>Error Analysis Prompting Enables Human-Like Translation Evaluation in Large Language Models</RefTitle>
        <RefYear>2024</RefYear>
        <RefBookTitle>Findings of the Association for Computational Linguistics: ACL 2024</RefBookTitle>
        <RefPage>8801-16</RefPage>
        <RefTotal>Lu Q, Qiu B, Ding L, Zhang K, Kocmi T, Tao D. Error Analysis Prompting Enables Human-Like Translation Evaluation in Large Language Models. In: Ku LW, Martins A, Srikumar V, editors. Findings of the Association for Computational Linguistics: ACL 2024. Association for Computational Linguistics; 2024. p. 8801-16. DOI: 10.18653&#47;v1&#47;2024.findings-acl.520</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.18653&#47;v1&#47;2024.findings-acl.520</RefLink>
      </Reference>
      <Reference refNo="43">
        <RefAuthor>Chatzikoumi E</RefAuthor>
        <RefTitle>How to evaluate machine translation: A review of automated and human metrics</RefTitle>
        <RefYear>2020</RefYear>
        <RefJournal>Natural Language Engineering</RefJournal>
        <RefPage>137-61</RefPage>
        <RefTotal>Chatzikoumi E. How to evaluate machine translation: A review of automated and human metrics. Natural Language Engineering. 2020 Mar;26(2):137-61. DOI: 10.1017&#47;S1351324919000469</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1017&#47;S1351324919000469</RefLink>
      </Reference>
      <Reference refNo="44">
        <RefAuthor>DiMascio C</RefAuthor>
        <RefTitle></RefTitle>
        <RefYear></RefYear>
        <RefBookTitle>cdimascio&#47;py-readability-metrics</RefBookTitle>
        <RefPage></RefPage>
        <RefTotal>DiMascio C. cdimascio&#47;py-readability-metrics. Github; &#91;last accessed 2026 Aug 20&#93;. Available from: https:&#47;&#47;github.com&#47;cdimascio&#47;py-readability-metrics</RefTotal>
        <RefLink>https:&#47;&#47;github.com&#47;cdimascio&#47;py-readability-metrics</RefLink>
      </Reference>
      <Reference refNo="45">
        <RefAuthor>Chan YH</RefAuthor>
        <RefTitle>Biostatistics 104: correlational analysis</RefTitle>
        <RefYear>2003</RefYear>
        <RefJournal>Singapore Medical Journal</RefJournal>
        <RefPage>614-9</RefPage>
        <RefTotal>Chan YH. Biostatistics 104: correlational analysis. Singapore Medical Journal. 2003 Dec;44(12):614-9.</RefTotal>
      </Reference>
      <Reference refNo="46">
        <RefAuthor>Akoglu H</RefAuthor>
        <RefTitle>User&#8217;s guide to correlation coefficients</RefTitle>
        <RefYear>2018</RefYear>
        <RefJournal>Turk J Emerg Med</RefJournal>
        <RefPage>91-3</RefPage>
        <RefTotal>Akoglu H. User&#8217;s guide to correlation coefficients. Turk J Emerg Med. 2018 Sep;18(3):91-3. DOI: 10.1016&#47;j.tjem.2018.08.00</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.1016&#47;j.tjem.2018.08.00</RefLink>
      </Reference>
      <Reference refNo="22">
        <RefAuthor>Reichenpfader D</RefAuthor>
        <RefAuthor>Dennst&#228;dt F</RefAuthor>
        <RefTitle></RefTitle>
        <RefYear>2026</RefYear>
        <RefBookTitle>RAT: Retrieval-augmented translation for medical texts (source code)</RefBookTitle>
        <RefPage></RefPage>
        <RefTotal>Reichenpfader D, Dennst&#228;dt F. RAT: Retrieval-augmented translation for medical texts (source code). Zenodo; 2026. 
DOI: 10.5281&#47;zenodo.21995221</RefTotal>
        <RefLink>https:&#47;&#47;doi.org&#47;10.5281&#47;zenodo.21995221</RefLink>
      </Reference>
    </References>
    <Media>
      <Tables>
        <Table format="png">
          <MediaNo>1</MediaNo>
          <MediaID>1</MediaID>
          <Caption><Pgraph><Mark1>Table 1: Average COMET translation scores (x&#175;) across model configurations, shown with (K) and without (NK) keyword conditioning (n&#61;500). Best performance highlighted in bold.</Mark1></Pgraph></Caption>
        </Table>
        <NoOfTables>1</NoOfTables>
      </Tables>
      <Figures>
        <Figure width="1606" height="1035" format="png">
          <MediaNo>1</MediaNo>
          <MediaID>1</MediaID>
          <Caption><Pgraph><Mark1>Figure 1: Overview of the S-RAT retrieval pipeline. Clinical concepts are extracted from the source report, mapped to SNOMED CT concepts, translated using the German SNOMED CT edition (with optional fallback dictionary), and inserted into the prompt to guide the LLM during translation.</Mark1></Pgraph></Caption>
        </Figure>
        <Figure width="765" height="669" format="png">
          <MediaNo>2</MediaNo>
          <MediaID>2</MediaID>
          <Caption><Pgraph><Mark1>Figure 2: Relationship between human preference scores and COMET-src score differences (S-RAT minus standard prompting). Positive x-values indicate that COMET-src favoured S-RAT, whereas positive y-values indicate that the human expert preferred the S-RAT translation.</Mark1></Pgraph></Caption>
        </Figure>
        <NoOfPictures>2</NoOfPictures>
      </Figures>
      <InlineFigures>
        <Figure width="103" height="43" format="png">
          <MediaNo>1</MediaNo>
          <MediaID>1</MediaID>
          <AltText>equation 1</AltText>
        </Figure>
        <NoOfPictures>1</NoOfPictures>
      </InlineFigures>
      <Attachments>
        <Attachment>
          <MediaNo>1</MediaNo>
          <MediaID mimeType="application/pdf" size="227194" filename="mibe000313.a1.pdf" url="" origFilename="Attachment1&#95;mibe000313.pdf">1</MediaID>
          <AttachmentTitle>Appendices A&#8211;G</AttachmentTitle>
        </Attachment>
        <NoOfAttachments>1</NoOfAttachments>
      </Attachments>
    </Media>
  </OrigData>
</GmsArticle>