<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR AI</journal-id><journal-id journal-id-type="publisher-id">ai</journal-id><journal-id journal-id-type="index">41</journal-id><journal-title>JMIR AI</journal-title><abbrev-journal-title>JMIR AI</abbrev-journal-title><issn pub-type="epub">2817-1705</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v5i1e78485</article-id><article-id pub-id-type="doi">10.2196/78485</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Benchmarking AI-Powered Translation of the EQ-5D-5L Patient-Reported Outcome Measure Using Automated Metrics: Comparative Evaluation Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Vashisht</surname><given-names>Himanshu</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ward</surname><given-names>Tom&#x00E1;s</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Muehlhausen</surname><given-names>Willie</given-names></name><degrees>DVM</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref></contrib></contrib-group><aff id="aff1"><institution>School of Computing, Dublin City University</institution><addr-line>Collins Ave Ext, Whitehall</addr-line><addr-line>Dublin</addr-line><country>Ireland</country></aff><aff id="aff2"><institution>Rinn Artificial Intelligence, Dublin City University</institution><addr-line>Dublin</addr-line><country>Ireland</country></aff><aff id="aff3"><institution>SAFIRA Clinical Research Ltd</institution><addr-line>Tipperary</addr-line><country>Ireland</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Mitra</surname><given-names>Avijit</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Byrom</surname><given-names>Bill</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Briva-Iglesias</surname><given-names>Vicent</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Himanshu Vashisht, MSc, School of Computing, Dublin City University, Collins Ave Ext, Whitehall, Dublin, Ireland, 353 899586131; <email>himanshu.vashisht3@mail.dcu.ie</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>4</day><month>8</month><year>2026</year></pub-date><volume>5</volume><elocation-id>e78485</elocation-id><history><date date-type="received"><day>03</day><month>06</month><year>2025</year></date><date date-type="rev-recd"><day>29</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>29</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Himanshu Vashisht, Tom&#x00E1;s Ward, Willie Muehlhausen. Originally published in JMIR AI (<ext-link ext-link-type="uri" xlink:href="https://ai.jmir.org">https://ai.jmir.org</ext-link>), 4.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR AI, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.ai.jmir.org/">https://www.ai.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://ai.jmir.org/2026/1/e78485"/><abstract><sec><title>Background</title><p>Patient-reported outcome measures (PROMs) are central to multinational clinical research, but high-quality translation and linguistic validation remain resource-intensive. AI-powered translation may accelerate this process, but its performance relative to validated human PROM translations requires systematic evaluation.</p></sec><sec><title>Objective</title><p>This benchmarking study evaluated the quality and comparability of 4 AI-powered translation services for the EuroQol 5-dimension 5-level (EQ-5D-5L) across 5 target languages, using official, linguistically validated human translations as the reference standard (gold standard).</p></sec><sec sec-type="methods"><title>Methods</title><p>The 43 text segments of the EQ-5D-5L were translated from English into Danish, Dutch, French, German, and Spanish using Google Translate, GPT-4.1, Amazon Translate, and DeepL. GPT-4.1 was evaluated with a structured medical-translator prompt, whereas Google Translate, Amazon Translate, and DeepL were evaluated using standard unprompted application programming interfaces without domain-specific glossary constraints. Outputs were benchmarked against official, validated human translations using 4 automated metrics: BLEU (bilingual evaluation understudy), METEOR (metric for evaluation of translation with explicit ordering), COMET (cross-lingual optimized metric for evaluation of translation), and BLEURT (bilingual evaluation understudy with representations from transformers). Friedman tests were used to assess overall between-service differences within each metric-language combination. When the Friedman test was significant, paired Wilcoxon signed-rank post hoc tests with Holm-Bonferroni correction were conducted. Descriptive summaries, score distributions, and sentence-level hotspot analyses were used to evaluate semantic similarity patterns and identify localized low-scoring deviations.</p></sec><sec sec-type="results"><title>Results</title><p>Friedman tests assessed whether the AI services differed in performance, whereas descriptive summaries and visualizations were used to determine whether scores clustered in ranges consistent with strong semantic similarity to the gold standard. Friedman tests identified statistically significant between-service differences in 11 of the 20 (55%; <italic>P</italic>&#x003C;.05) metric-language combinations. Subsequent paired Wilcoxon signed-rank post hoc tests with Holm-Bonferroni correction identified 11 significant pairwise differences, with adjusted <italic>P</italic> values ranging from &#x003C;.001 to .049. Most of these differences were detected by surface-overlap metrics (10/11 for BLEU or METEOR), whereas only 1 of 11 was detected by a semantic metric (BLEURT), suggesting that many between-service differences were stylistic rather than meaning-altering. Descriptive and visual analyses further showed that semantic similarity was generally high across services, while low-scoring deviations clustered in specific linguistic hotspots, particularly domain headers, abstract health concepts, and short, context-dependent interface strings.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Among the evaluated high-resource European languages, AI translation services showed high semantic similarity to the validated human translations, although localized conceptual deviations persisted. These findings suggest that AI can support the generation of translations for PROM workflows in these languages; however, expert human review may still be required to confirm conceptual equivalence. The practical relevance of isolated header differences could not be assessed in the present study, whereas abstract health concepts and other clinically sensitive phrasing should be evaluated in further research.</p></sec></abstract><kwd-group><kwd>patient-reported outcomes</kwd><kwd>electronic clinical outcome assessment</kwd><kwd>machine translation</kwd><kwd>EQ-5D-5L</kwd><kwd>translation quality</kwd><kwd>linguistic validation</kwd><kwd>clinical research</kwd><kwd>multilingual studies</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Patient-reported outcome measures (PROMs) are pivotal in clinical research because they capture health status, symptoms, functioning, and quality of life directly from patients. The increasing global reach of clinical trials necessitates that PROMs be accurately translated and culturally adapted for diverse linguistic and cultural contexts. Traditionally, this process has involved a rigorous linguistic validation methodology, typically including forward and backward-translation, cognitive debriefing, and expert review, ensuring conceptual, item, and measurement equivalence across languages [<xref ref-type="bibr" rid="ref1">1</xref>]. This comprehensive approach, while robust, is resource-intensive and time-consuming, posing challenges for the efficient deployment of PROMs in multinational studies [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>].</p></sec><sec id="s1-2"><title>Challenges in Traditional Translation</title><p>The traditional linguistic validation process, while crucial for ensuring high-quality PROM translations, faces several inherent challenges. These include the significant time and financial resources required to execute rigorous forward-backward translation, cognitive debriefing, and expert reconciliation processes [<xref ref-type="bibr" rid="ref4">4</xref>]. The complexity of managing these multistep workflows across numerous languages can lead to delays in clinical trial timelines, impacting the overall efficiency and cost-effectiveness of global research [<xref ref-type="bibr" rid="ref5">5</xref>]. Furthermore, the availability of qualified human translators with expertise in both medical terminology and specific cultural nuances can be limited, particularly for less commonly spoken languages or highly specialized therapeutic areas [<xref ref-type="bibr" rid="ref6">6</xref>].</p></sec><sec id="s1-3"><title>Emergence of AI in Translation</title><p>The rapid advancements in AI, particularly in natural language processing and, more recently, large language models (LLMs), have introduced new possibilities for automated and semiautomated translation. AI-powered translation services offer the potential to significantly streamline the translation process, reduce costs, and accelerate the availability of PROMs in multiple languages [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. These technologies leverage vast datasets and sophisticated algorithms to generate translations with remarkable fluency and, in many cases, accuracy comparable to that of human translation for general texts [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. The integration of AI tools could potentially alleviate some of the burdens associated with traditional translation workflows, making multilingual electronic clinical outcome assessment deployment more feasible and efficient [<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref12">12</xref>]. Recent PROM-specific evidence also suggests that LLMs can produce clinically relevant translation outputs under certain conditions, although performance remains dependent on the instrument, language pair, and evaluation framework [<xref ref-type="bibr" rid="ref13">13</xref>]. However, recent health care research has emphasized that LLMs should be evaluated and deployed with careful attention to oversight, transparency, and task-specific risk, particularly in clinically consequential settings [<xref ref-type="bibr" rid="ref14">14</xref>].</p></sec><sec id="s1-4"><title>Overview of AI Translation Models</title><p>Current AI translation tools vary in architecture and controllability. Google Translate, Amazon Translate, and DeepL are widely used neural machine translation (NMT) services, whereas GPT-4.1 represents a newer LLM-based approach that can be guided through explicit prompting [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref15">15</xref>-<xref ref-type="bibr" rid="ref17">17</xref>]. Understanding how these systems compare with industry-standard clinical translations (linguistic validation) is important before they can be used responsibly in sensitive settings such as PROM translation. Recent research suggests that machine translation in the era of LLMs has improved substantially, especially in high-resource languages, while still exhibiting limitations that may remain relevant for specialized applications such as clinical questionnaires [<xref ref-type="bibr" rid="ref18">18</xref>].</p></sec><sec id="s1-5"><title>Specific Characteristics of EQ-5D-5L for Translation</title><p>The EuroQol 5-dimension 5-level (EQ-5D-5L) is a widely used generic PROM for assessing health-related quality of life across various populations and diseases [<xref ref-type="bibr" rid="ref19">19</xref>]. Its standardized, simple, and direct language structure, consisting of 5 dimensions (mobility, self-care, usual activities, pain or discomfort, and anxiety or depression), each with 5 severity levels, makes it seem straightforward to translate. However, ensuring the linguistic integrity and measurement equivalence of its translations is critical for the validity of data collected in global trials [<xref ref-type="bibr" rid="ref20">20</xref>,<xref ref-type="bibr" rid="ref21">21</xref>]. Ensuring conceptual equivalence is, therefore, essential to preserve the instrument&#x2019;s validity in multinational research.</p></sec><sec id="s1-6"><title>Goal of This Study</title><p>This study aimed to benchmark (via algorithms) the quality and comparability of AI-generated EQ-5D-5L translations against linguistically validated human translations. It was designed as a practical benchmarking study against an established clinical translation standard rather than as a test of full linguistic validation or zero-shot performance on unseen proprietary text.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>This study used a comparative benchmarking design to assess the quality and comparability of AI-powered translation outputs of the EQ-5D-5L against official, linguistically validated human translations. The methodology involved translating the standardized English version of the EQ-5D-5L into 5 target high-resource target languages: Danish (DA), Dutch (NL), French (FR), German (DE), and Spanish (ES), using 4 distinct AI translation services. <xref ref-type="fig" rid="figure1">Figure 1</xref> illustrates the comprehensive workflow used in this study, from the initial document translation to the final quality evaluation and analysis.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Study workflow for this methodological benchmarking study of AI-powered EQ-5D-5L translation. The 43 UK English EQ-5D-5L source segments were translated into 5 target languages (Danish, Dutch, French, German, and Spanish) using 4 AI services (Google Translate, Amazon Translate, DeepL, and GPT-4.1). AI-generated forward translations were then benchmarked against official, linguistically validated human translations using 4 automated metrics (BLEU, METEOR, COMET, and BLEURT). The resulting scores were summarized using descriptive statistics, qualitative outlier analysis, and inferential testing (Friedman tests with Kendall <italic>W</italic> and Holm-Bonferroni-adjusted Wilcoxon post hoc comparisons), followed by visualization of score distributions and hotspot patterns. BLEU: bilingual evaluation understudy; BLEURT: bilingual evaluation understudy with representations from transformers; COMET: cross-lingual optimized metric for evaluation of translation; EN: English; EQ-5D-5L: EuroQol 5-dimension 5-level; METEOR: metric for evaluation of translation with explicit ordering.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e78485_fig01.png"/></fig></sec><sec id="s2-2"><title>Translation Services Used</title><p>Four leading AI translation services were selected for this study:</p><list list-type="bullet"><list-item><p>Google Translate: This is a widely used NMT service known for its extensive language support and continuous advancements [<xref ref-type="bibr" rid="ref10">10</xref>].</p></list-item><list-item><p>GPT-4.1 (OpenAI): This is a state-of-the-art LLM [<xref ref-type="bibr" rid="ref17">17</xref>]. For this study, translations were generated through the OpenAI application programming interface (API) using the GPT-4.1 snapshot identifier, gpt-4.1-2025-04-14. To ensure reproducibility, fixed settings of a seed of 42 and a temperature of 0 were used. A detailed system prompt was used to instruct the model to act as a professional medical translator, emphasizing accuracy, clinical context, and fidelity to the source text (see <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> for the full prompt). No equivalent prompt-based configuration was available in the standard workflows used by the NMT services.</p></list-item><list-item><p>Amazon Translate (Amazon Web Services [AWS]): This is an NMT service offering high scalability and integration with other cloud services [<xref ref-type="bibr" rid="ref16">16</xref>].</p></list-item><list-item><p>DeepL: This is an NMT service renowned for its high-quality, nuanced translations, particularly for European languages, leveraging advanced deep learning architectures [<xref ref-type="bibr" rid="ref15">15</xref>].</p></list-item></list><p>These services represent a diverse set of current AI translation technologies. No equivalent domain-specific prompt, medical glossary, or custom terminology constraint was applied to Google Translate, Amazon Translate, or DeepL. These services were evaluated using their standard API translation behavior, as commercially available to end users at the time of analysis. Because the evaluated systems differ in interface and controllability, input standardization was necessarily imperfect. GPT-4.1 permitted explicit instruction through a system prompt, whereas the NMT services did not provide an equivalent prompt-based mechanism within the workflow used here. Therefore, this study reflects a realistic applied comparison of available translation services rather than a strictly parameter-matched head-to-head experiment.</p></sec><sec id="s2-3"><title>Provenance and Reproducibility</title><p>To address potential model drift and ensure reproducibility, all AI translations were generated within a fixed time window in August 2025. The specific service versions and end points used were as follows:</p><list list-type="bullet"><list-item><p>Google Translate: Requests were submitted through the Google Cloud Translation API (version 3) end point</p></list-item><list-item><p>DeepL: Requests were made to the DeepL API (version 2) end point</p></list-item><list-item><p>Amazon Translate: Requests were made through the AWS software development kit for Python (boto3; version 1.40.66), targeting the eu-west-1 region</p></list-item><list-item><p>GPT-4.1: As detailed previously, all requests used the GPT-4.1 API snapshot gpt-4.1-2025-04-14, with a fixed seed of 42 and a temperature of 0</p></list-item></list><p>The access dates reported for API documentation references indicate when the documentation sources were consulted and do not necessarily correspond to the dates on which translation outputs were generated.</p></sec><sec id="s2-4"><title>Data Collection and Preparation</title><p>The source text consisted of the 43 unique text segments from the standard UK English version of the EQ-5D-5L, comprising the instrument&#x2019;s 5 dimensions and the EQ visual analog scale instructions. For interpretive transparency, <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> presents annotated screenshots of the corresponding sample UK English digital questionnaire structure from the EuroQol user guide. These text segments were systematically translated into the 5 target languages using each of the 4 AI translation services. Official, linguistically validated human translations for each target language were obtained from the EuroQol Research Foundation and served as the gold standard reference for benchmarking. These translations were developed through a multistep process, including forward-backward translation and cognitive debriefing, and represent the current industry standard for clinical use.</p><p>Sentence-level translation was used because the workflow compared each source segment with its aligned gold standard counterpart, and many EQ-5D-5L elements function as discrete response or instruction units rather than cohesive prose. However, this design removed the broader questionnaire context that some systems might otherwise use to improve lexical consistency, disambiguation, and register, and it provided only an approximation of dependence among related instrument segments.</p></sec><sec id="s2-5"><title>Automated Translation Quality Metrics</title><p>Evaluating the quality of machine-translated PROMs requires robust and reliable metrics. This study uses 4 widely recognized automated machine translation metrics. All scores were computed at the sentence level.</p><list list-type="bullet"><list-item><p>BLEU (bilingual evaluation understudy): BLEU is a precision-based metric that measures <italic>n</italic>-gram (contiguous sequences of <italic>n</italic> words) overlap between the candidate and reference translations [<xref ref-type="bibr" rid="ref22">22</xref>]. We used sacrebleu (version 2.5.0) with the &#x201C;13a&#x201D; tokenizer and normalized scoring. BLEU scores range from 0 to 1, with 1 indicating a perfect match to the reference. Generally, BLEU scores &#x003E;0.5 reflect high-quality, fluent translations, whereas scores &#x003C;0.3 often indicate poor correlation with human references.</p></list-item><list-item><p>METEOR (metric for evaluation of translation with explicit ordering): METEOR is a recall-oriented metric that considers unigram (single word) matching, stemming, and synonym matching [<xref ref-type="bibr" rid="ref23">23</xref>]. Scores were computed using Natural Language Toolkit (version 3.9.1). METEOR scores range from 0 to 1, with higher scores indicating better translation quality. Similar to BLEU, scores approaching 1.0 indicate high quality, whereas lower scores suggest significant lexical deviation.</p></list-item><list-item><p>COMET (cross-lingual optimization for machine translation evaluation): COMET is a neural framework that uses a pretrained cross-lingual encoder to produce more robust evaluations by assessing semantic similarity [<xref ref-type="bibr" rid="ref24">24</xref>]. We used the Unbabel/wmt22-comet-da reference-based model, which has shown a high correlation with human judgments. COMET scores typically range from &#x2212;1 to 1, with 1 representing a perfect translation. While no universal threshold exists, scores &#x2265;0.80 typically indicate strong semantic equivalence, whereas scores &#x003C;0.60 often signal potential semantic errors requiring review.</p></list-item><list-item><p>BLEURT (bilingual evaluation understudy representation-enhanced): BLEURT is a neural metric trained to predict human judgments of translation quality [<xref ref-type="bibr" rid="ref25">25</xref>]. We used the official BLEURT-20 checkpoint. BLEURT scores generally range from &#x2212;1 to 1, with higher scores indicating better quality. Consistent with COMET, scores approaching 1.0 reflect high semantic fidelity.</p></list-item></list><p>BLEU and METEOR are more sensitive to lexical and word-order variation, whereas COMET and BLEURT are more informative for semantic adequacy. Because COMET and BLEURT were used in a reference-based form against the gold standard translations, they may reward closer alignment with the validated reference wording and penalize clinically acceptable alternative phrasings. Recent work has also aimed to improve the interpretability of neural translation metrics through fine-grained error detection, reinforcing the importance of semantic evaluation approaches that move beyond simple lexical overlap [<xref ref-type="bibr" rid="ref26">26</xref>].</p><p>To aid interpretation, we also summarized the proportion of sentence-level scores falling within descriptive quality bands. Because no universally accepted pass or fail thresholds exist for applying these automated metrics to PROM translation benchmarking, the bands were used only as pragmatic interpretive aids for summarizing score distributions. For BLEU and METEOR, which are more sensitive to lexical and word-order overlap, scores &#x2265;0.50 were categorized as high, scores from 0.30 to &#x003C;0.50 as mid, and scores &#x003C;0.30 as low. For COMET and BLEURT, which were used as semantic similarity metrics, scores &#x2265;0.80 were categorized as high, scores from 0.60 to &#x003C;0.80 as mid, and scores &#x003C;0.60 as low. These bands were not treated as formally validated pass or fail criteria for PROM translation quality. These metrics provide objective, quantitative measures of translation quality that are crucial for benchmarking different AI services.</p></sec><sec id="s2-6"><title>Data Analysis</title><sec id="s2-6-1"><title>Overview of Analyses</title><p>The analyses addressed 2 complementary questions. First, inferential comparisons were used to test whether the 4 AI services differed in performance within each metric-language combination. Second, descriptive summaries, violin plots, and sentence-level heatmaps were used to evaluate whether outputs were generally concentrated in ranges consistent with acceptable semantic similarity to the validated reference translations.</p></sec><sec id="s2-6-2"><title>Descriptive Statistics</title><p>For each AI service within each metric-language combination, we calculated the median and IQR across the 43 source segments to summarize the central tendency and dispersion of sentence-level translation quality scores.</p></sec><sec id="s2-6-3"><title>Inferential Statistics and Effect Sizes</title><p>Given the repeated-measures design, Friedman tests were used to compare the score distributions among the 4 AI services within each metric-language combination. To quantify the magnitude of any significant findings, Kendall <italic>W</italic> was calculated as the omnibus effect size. When the Friedman test was significant (<italic>P</italic>&#x003C;.05), paired Wilcoxon signed-rank post hoc tests were performed with Holm-Bonferroni correction, and the pairwise effect size <italic>r</italic> was calculated for each significant contrast. Because the Friedman and Wilcoxon signed-rank post hoc tests address different hypotheses, omnibus Friedman <italic>P</italic> values were interpreted separately from Holm-Bonferroni&#x2013;adjusted pairwise <italic>P</italic> values. Kendall <italic>W</italic> and the pairwise Wilcoxon effect size <italic>r</italic> quantify different levels of comparison and should not be interpreted as directly interchangeable. Kendall <italic>W</italic> summarizes the overall degree of separation among all 4 services within a metric-language combination, whereas <italic>r</italic> quantifies the magnitude of a specific pairwise contrast.</p></sec><sec id="s2-6-4"><title>Qualitative Outlier Analysis</title><p>Finally, to supplement the inferential and descriptive analyses, a qualitative outlier analysis was conducted to identify localized failure modes. For each metric-language combination, sentence-level scores at or below the 5th percentile were flagged as potential hotspots and aggregated across services, languages, and metrics to identify recurring low-scoring segments. As a descriptive sensitivity analysis, we also summarized the total number of flagged instances generated by each metric across all language-service combinations to compare how frequently each metric identified potential problem segments. This comparison was descriptive and was intended to characterize relative filtering sensitivity rather than to provide a separate inferential test.</p><p>Flagged discrepancies were then reviewed qualitatively and assigned to prespecified linguistic error categories. Conceptual errors referred to translations that failed to convey the intended clinical construct represented in the gold standard. Semantic errors referred to meaning shifts that altered the sense of the item without necessarily changing its conceptual domain. Intensity errors referred to translations that distorted the severity or strength of the source wording. Domain errors referred to translations that shifted a term into an inappropriate contextual field or usage domain. These categories were used as an interpretive framework for describing representative outliers rather than as a formally validated taxonomy. Categorization was informed by comparison with the gold standard translations and by exploratory triangulation using informal native-speaker discussions, web-based language resources, and LLMs.</p></sec></sec><sec id="s2-7"><title>Ethical Considerations</title><p>This study was a methodological benchmarking analysis and did not involve human participants, patient-level data, or identifiable personal information. The materials analyzed consisted of EQ-5D-5L source-text segments and official, linguistically validated translations used as benchmark references for translation-quality evaluation. Because no human-participant research procedures were conducted and no identifiable human data were analyzed, institutional review board review was not required. Informed consent was not applicable. No compensation was provided. No identifying images of participants or users were included in the manuscript or supplementary materials.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Overview of Analyses</title><p>The results are presented in 2 complementary parts. First, inferential analyses were used to test whether the 4 AI translation services differed from one another within each metric-language combination. Second, descriptive summaries and visual analyses were used to assess the distribution of translation quality scores relative to the validated human translations and to identify localized sentence-level hotspots.</p></sec><sec id="s3-2"><title>Statistical Comparison of AI Services</title><p>Translation quality scores from the 4 AI services (Google, GPT-4.1, Amazon, and DeepL) were compared across all 20 metric-language combinations. The Friedman tests revealed statistically significant differences in 11 of the 20 (55%) combinations, indicating that the performance distributions of the services were not identical.</p><p>To determine which specific services differed, paired Wilcoxon signed-rank post hoc tests with a Holm-Bonferroni correction were performed on the 11 significant Friedman test results. This analysis confirmed 11 statistically significant pairwise differences, with adjusted <italic>P</italic> values ranging from &#x003C;.001 to .049. This is the central finding: the AI services are not functionally interchangeable, and statistically significant performance differences depended on both the target language and the evaluation metric.</p></sec><sec id="s3-3"><title>Descriptive and Inferential Findings With Effect Sizes</title><p>The significant differences clustered around specific metrics (<xref ref-type="table" rid="table1">Table 1</xref>). Of the 11 significant pairwise differences, 10 (91%) were identified by the surface-overlap metrics (BLEU and METEOR), which measure stylistic and lexical similarity. In contrast, the semantic metrics (COMET and BLEURT), which measure meaning, together only accounted for 1 of the 11 differences. Taken together, these findings suggest that the services differ more in stylistic realization than in meaning preservation. To visualize this semantic comparability directly, <xref ref-type="fig" rid="figure2">Figure 2</xref> presents COMET score distributions, as COMET is a reference-based semantic metric with a strong reported correlation with human judgments. Individual score distributions for each target language are detailed in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><p>The post hoc summary shows Holm-Bonferroni&#x2013;adjusted <italic>P</italic> values and pairwise effect size <italic>r</italic> from paired Wilcoxon signed-rank tests for significant pairwise comparisons only. Friedman <italic>P</italic> values and post hoc adjusted <italic>P</italic> values address different levels of comparison and are, therefore, not directly comparable: the Friedman test evaluates an omnibus difference across all 4 services, whereas each Wilcoxon signed-rank test evaluates a specific service pair.</p><p>This interpretation was supported by the effect size analysis. Omnibus effect sizes were generally small, with a median Kendall <italic>W</italic> of 0.08 (IQR 0.07-0.10) across the 11 significant tests, indicating limited overall separation among all 4 AI translation services within a metric-language combination. In contrast, some post hoc pairwise effect sizes were moderate to large, showing that specific service pairs could differ meaningfully even when the overall omnibus separation remained modest.</p><p>For example, in the BLEU score for Spanish, Google (median 0.325, IQR 0.199-0.632) performed significantly worse than Amazon (median 0.425, IQR 0.230-1.000; adjusted <italic>P</italic>=.001; <italic>r</italic>=0.75). Conversely, for the semantic metric COMET in Dutch, the Friedman test indicated an overall difference among services, but no pairwise comparison remained significant after Holm-Bonferroni correction. This finding reinforces the interpretation that semantic metric differences were limited and that most significant pairwise differences were concentrated in surface-overlap metrics.</p><p>Threshold-based descriptive summaries further supported this interpretation. Across languages and services, most COMET and BLEURT scores fell within ranges consistent with high semantic similarity to the validated reference translations, whereas BLEU and METEOR showed a greater proportion of lower scores because of their sensitivity to lexical and word-order variation. Detailed threshold-based summaries by metric, language, and AI translation service are provided in Table S1 in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Descriptive statistics and inferential comparison results for this methodological benchmarking study of AI-powered EQ-5D-5L translation<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Metric</td><td align="left" valign="bottom">Language (code)</td><td align="left" valign="bottom">Chi-square (<italic>df</italic>); Friedman test</td><td align="left" valign="bottom"><italic>P</italic> value</td><td align="left" valign="bottom">Kendall <italic>W</italic><sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="bottom">Google, median (IQR)</td><td align="left" valign="bottom">GPT-4.1, median (IQR)</td><td align="left" valign="bottom">Amazon, median (IQR)</td><td align="left" valign="bottom">DeepL, median (IQR)</td><td align="left" valign="bottom">Post hoc summary (adjusted <italic>P</italic> value; <italic>r</italic>)</td></tr></thead><tbody><tr><td align="left" valign="top">COMET<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">Danish (DA)</td><td align="left" valign="top">5.036 (3)</td><td align="left" valign="top">.17</td><td align="left" valign="top">0.04</td><td align="left" valign="top">0.941 (0.868-0.981)</td><td align="left" valign="top">0.938 (0.876-0.977)</td><td align="left" valign="top">0.941 (0.871-0.976)</td><td align="left" valign="top">0.954 (0.890-0.980)</td><td align="left" valign="top">No significant difference</td></tr><tr><td align="left" valign="top">COMET</td><td align="left" valign="top">German (DE)</td><td align="left" valign="top">8.118 (3)</td><td align="left" valign="top">.04</td><td align="left" valign="top">0.06</td><td align="left" valign="top">0.920 (0.720-0.949)</td><td align="left" valign="top">0.905 (0.779-0.930)</td><td align="left" valign="top">0.852 (0.720-0.941)</td><td align="left" valign="top">0.909 (0.734-0.933)</td><td align="left" valign="top">No pairwise differences after correction</td></tr><tr><td align="left" valign="top">COMET</td><td align="left" valign="top">French (FR)</td><td align="left" valign="top">6.01 (3)</td><td align="left" valign="top">.11</td><td align="left" valign="top">0.05</td><td align="left" valign="top">0.905 (0.789-0.947)</td><td align="left" valign="top">0.912 (0.796-0.950)</td><td align="left" valign="top">0.904 (0.789-0.942)</td><td align="left" valign="top">0.912 (0.836-0.936)</td><td align="left" valign="top">No significant difference</td></tr><tr><td align="left" valign="top">COMET</td><td align="left" valign="top">Dutch (NL)</td><td align="left" valign="top">9.296 (3)</td><td align="left" valign="top">.03</td><td align="left" valign="top">0.07</td><td align="left" valign="top">0.933 (0.901-0.954)</td><td align="left" valign="top">0.947 (0.921-0.968)</td><td align="left" valign="top">0.937 (0.899-0.963)</td><td align="left" valign="top">0.945 (0.907-0.964)</td><td align="left" valign="top">No pairwise differences after correction</td></tr><tr><td align="left" valign="top">COMET</td><td align="left" valign="top">Spanish (ES)</td><td align="left" valign="top">7.56 (3)</td><td align="left" valign="top">.06</td><td align="left" valign="top">0.06</td><td align="left" valign="top">0.943 (0.803-0.962)</td><td align="left" valign="top">0.926 (0.837-0.962)</td><td align="left" valign="top">0.945 (0.848-0.971)</td><td align="left" valign="top">0.944 (0.829-0.966)</td><td align="left" valign="top">No significant difference</td></tr><tr><td align="left" valign="top">BLEU<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">Danish (DA)</td><td align="left" valign="top">1.11 (3)</td><td align="left" valign="top">.78</td><td align="left" valign="top">0.009</td><td align="left" valign="top">0.707 (0.294-1.000)</td><td align="left" valign="top">0.569 (0.301-1.000)</td><td align="left" valign="top">0.502 (0.230-0.867)</td><td align="left" valign="top">0.649 (0.230-1.000)</td><td align="left" valign="top">No significant difference</td></tr><tr><td align="left" valign="top">BLEU</td><td align="left" valign="top">German (DE)</td><td align="left" valign="top">10.478 (3)</td><td align="left" valign="top">.02</td><td align="left" valign="top">0.08</td><td align="left" valign="top">0.508 (0.147-0.760)</td><td align="left" valign="top">0.393 (0.129-0.702)</td><td align="left" valign="top">0.323 (0.086-0.695)</td><td align="left" valign="top">0.446 (0.162-0.718)</td><td align="left" valign="top">Google vs GPT-4.1 (adjusted <italic>P</italic>=.03; <italic>r</italic>=0.54)</td></tr><tr><td align="left" valign="top">BLEU</td><td align="left" valign="top">French (FR)</td><td align="left" valign="top">12.551 (3)</td><td align="left" valign="top">.006</td><td align="left" valign="top">0.10</td><td align="left" valign="top">0.102 (0.054-0.304)</td><td align="left" valign="top">0.222 (0.066-0.361)</td><td align="left" valign="top">0.095 (0.054-0.261)</td><td align="left" valign="top">0.161 (0.062-0.248)</td><td align="left" valign="top">GPT-4.1 vs Amazon (adjusted <italic>P</italic>=.004; <italic>r</italic>=0.66); Google vs GPT-4.1 (adjusted <italic>P</italic>=.01; <italic>r</italic>=0.62)</td></tr><tr><td align="left" valign="top">BLEU</td><td align="left" valign="top">Dutch (NL)</td><td align="left" valign="top">8.027 (3)</td><td align="left" valign="top">.047</td><td align="left" valign="top">0.06</td><td align="left" valign="top">0.382 (0.097-0.745)</td><td align="left" valign="top">0.597 (0.231-0.800)</td><td align="left" valign="top">0.322 (0.086-0.652)</td><td align="left" valign="top">0.395 (0.077-0.728)</td><td align="left" valign="top">GPT-4.1 vs Amazon (adjusted <italic>P</italic>=.02; <italic>r</italic>=0.63)</td></tr><tr><td align="left" valign="top">BLEU</td><td align="left" valign="top">Spanish (ES)</td><td align="left" valign="top">20.661 (3)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.16</td><td align="left" valign="top">0.325 (0.199-0.632)</td><td align="left" valign="top">0.425 (0.175-0.743)</td><td align="left" valign="top">0.425 (0.230-1.000)</td><td align="left" valign="top">0.398 (0.240-0.688)</td><td align="left" valign="top">Google vs Amazon (adjusted <italic>P</italic>=.001; <italic>r</italic>=0.75)</td></tr><tr><td align="left" valign="top">METEOR<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup></td><td align="left" valign="top">Danish (DA)</td><td align="left" valign="top">2.122 (3)</td><td align="left" valign="top">.55</td><td align="left" valign="top">0.02</td><td align="left" valign="top">0.837 (0.642-0.998)</td><td align="left" valign="top">0.837 (0.658-0.998)</td><td align="left" valign="top">0.807 (0.639-0.944)</td><td align="left" valign="top">0.865 (0.653-0.998)</td><td align="left" valign="top">No significant difference</td></tr><tr><td align="left" valign="top">METEOR</td><td align="left" valign="top">German (DE)</td><td align="left" valign="top">9.684 (3)</td><td align="left" valign="top">.02</td><td align="left" valign="top">0.08</td><td align="left" valign="top">0.769 (0.490-0.984)</td><td align="left" valign="top">0.711 (0.509-0.830)</td><td align="left" valign="top">0.598 (0.354-0.830)</td><td align="left" valign="top">0.778 (0.522-0.830)</td><td align="left" valign="top">No pairwise differences after correction</td></tr><tr><td align="left" valign="top">METEOR</td><td align="left" valign="top">French (FR)</td><td align="left" valign="top">17.166 (3)</td><td align="left" valign="top">&#x003C;.001</td><td align="left" valign="top">0.13</td><td align="left" valign="top">0.391 (0.264-0.591)</td><td align="left" valign="top">0.500 (0.347-0.753)</td><td align="left" valign="top">0.301 (0.223-0.516)</td><td align="left" valign="top">0.424 (0.343-0.517)</td><td align="left" valign="top">GPT-4.1 vs Amazon (adjusted <italic>P</italic>&#x003C;.001; <italic>r</italic>=0.81); Google vs GPT-4.1 (adjusted <italic>P</italic>=.005; <italic>r</italic>=0.68); GPT-4.1 vs DeepL (adjusted <italic>P</italic>=.049; <italic>r</italic>=0.44)</td></tr><tr><td align="left" valign="top">METEOR</td><td align="left" valign="top">Dutch (NL)</td><td align="left" valign="top">11.305 (3)</td><td align="left" valign="top">.01</td><td align="left" valign="top">0.09</td><td align="left" valign="top">0.747 (0.571-0.869)</td><td align="left" valign="top">0.830 (0.692-0.981)</td><td align="left" valign="top">0.701 (0.562-0.843)</td><td align="left" valign="top">0.736 (0.625-0.882)</td><td align="left" valign="top">GPT-4.1 vs Amazon (adjusted <italic>P</italic>&#x003C;.001; <italic>r</italic>=0.80); Google vs GPT-4.1 (adjusted <italic>P</italic>=.03; <italic>r</italic>=0.67)</td></tr><tr><td align="left" valign="top">METEOR</td><td align="left" valign="top">Spanish (ES)</td><td align="left" valign="top">11.795 (3)</td><td align="left" valign="top">.008</td><td align="left" valign="top">0.09</td><td align="left" valign="top">0.755 (0.708-0.963)</td><td align="left" valign="top">0.810 (0.724-0.976)</td><td align="left" valign="top">0.803 (0.661-0.996)</td><td align="left" valign="top">0.755 (0.628-0.981)</td><td align="left" valign="top">No pairwise differences after correction</td></tr><tr><td align="left" valign="top">BLEURT<sup><xref ref-type="table-fn" rid="table1fn6">f</xref></sup></td><td align="left" valign="top">Danish (DA)</td><td align="left" valign="top">5.798 (3)</td><td align="left" valign="top">.12</td><td align="left" valign="top">0.05</td><td align="left" valign="top">0.890 (0.756-0.964)</td><td align="left" valign="top">0.910 (0.805-0.959)</td><td align="left" valign="top">0.875 (0.762-0.938)</td><td align="left" valign="top">0.904 (0.811-0.960)</td><td align="left" valign="top">No significant difference</td></tr><tr><td align="left" valign="top">BLEURT</td><td align="left" valign="top">German (DE)</td><td align="left" valign="top">4.409 (3)</td><td align="left" valign="top">.22</td><td align="left" valign="top">0.03</td><td align="left" valign="top">0.874 (0.756-0.914)</td><td align="left" valign="top">0.834 (0.718-0.875)</td><td align="left" valign="top">0.819 (0.765-0.888)</td><td align="left" valign="top">0.842 (0.777-0.874)</td><td align="left" valign="top">No significant difference</td></tr><tr><td align="left" valign="top">BLEURT</td><td align="left" valign="top">French (FR)</td><td align="left" valign="top">8.737 (3)</td><td align="left" valign="top">.03</td><td align="left" valign="top">0.07</td><td align="left" valign="top">0.827 (0.727-0.896)</td><td align="left" valign="top">0.853 (0.759-0.903)</td><td align="left" valign="top">0.831 (0.733-0.900)</td><td align="left" valign="top">0.829 (0.763-0.872)</td><td align="left" valign="top">Google vs GPT-4.1 (adjusted <italic>P</italic>=.04; <italic>r</italic>=0.53)</td></tr><tr><td align="left" valign="top">BLEURT</td><td align="left" valign="top">Dutch (NL)</td><td align="left" valign="top">3.807 (3)</td><td align="left" valign="top">.28</td><td align="left" valign="top">0.03</td><td align="left" valign="top">0.852 (0.802-0.964)</td><td align="left" valign="top">0.905 (0.830-0.957)</td><td align="left" valign="top">0.856 (0.790-0.931)</td><td align="left" valign="top">0.890 (0.802-0.968)</td><td align="left" valign="top">No significant difference</td></tr><tr><td align="left" valign="top">BLEURT</td><td align="left" valign="top">Spanish (ES)</td><td align="left" valign="top">4.846 (3)</td><td align="left" valign="top">.18</td><td align="left" valign="top">0.04</td><td align="left" valign="top">0.905 (0.855-0.942)</td><td align="left" valign="top">0.930 (0.841-0.967)</td><td align="left" valign="top">0.930 (0.818-0.993)</td><td align="left" valign="top">0.909 (.841-0.971)</td><td align="left" valign="top">No significant difference</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>For each of the 20 metric-language combinations, the table reports sentence-level median and IQR (Q1-Q3) scores across 43 source-text segments for 4 AI services, together with Friedman test results, Kendall <italic>W</italic>, and Holm-adjusted Wilcoxon post hoc comparisons among AI services, based on scores computed against the gold standard translations.</p></fn><fn id="table1fn2"><p><sup>b</sup>Kendall <italic>W</italic> values are individual omnibus effect sizes from the corresponding Friedman tests and are not median values. The median and IQR ranges apply only to the AI service score columns.</p></fn><fn id="table1fn3"><p><sup>c</sup>COMET: cross-lingual optimized metric for evaluation of translation.</p></fn><fn id="table1fn4"><p><sup>d</sup>BLEU: bilingual evaluation understudy.</p></fn><fn id="table1fn5"><p><sup>e</sup>METEOR: metric for evaluation of translation with explicit ordering.</p></fn><fn id="table1fn6"><p><sup>f</sup>BLEURT: bilingual evaluation understudy with representations from transformers.</p></fn></table-wrap-foot></table-wrap><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Distribution of sentence-level COMET scores across 5 target languages for 4 AI translation services in this methodological benchmarking study of EQ-5D-5L translation. Each violin plot summarizes the distribution of semantic similarity scores for the 43 translated source-text segments benchmarked against the official, linguistically validated human translations (gold standard). The dashed horizontal line at COMET=0.80 marks the descriptive threshold used to indicate strong semantic similarity. Across languages, most scores cluster above this threshold, consistent with generally high semantic comparability despite localized sentence-level variation. COMET: cross-lingual optimized metric for evaluation of translation. The target languages are represented in the figure as follows: DA: Danish; NL: Dutch; FR: French; DE: German; and ES: Spanish. EQ-5D-5L: EuroQol 5-dimension 5-level.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e78485_fig02.png"/></fig></sec><sec id="s3-4"><title>Qualitative Outlier Analysis</title><p>To supplement the aggregate analyses, we conducted a qualitative outlier analysis by benchmarking AI outputs directly against the official, linguistically validated human translations (gold standard). Outliers were defined as sentence-level scores at or below the 5th percentile for each metric. This analysis was intended to localize failure modes rather than determine whether translation services were globally acceptable or unacceptable.</p><p>As a descriptive count-based sensitivity analysis, BLEU was the most discriminative metric for flagging potential low-scoring segments (158 instances), followed by METEOR (73), BLEURT (60), and COMET (60). Outlier counts were then examined across services. The distribution was relatively even, ranging from 85 flagged instances for DeepL to 92 for Amazon Translate, suggesting that all services showed a similar frequency of localized deviations from the gold standard.</p><p>Outliers were not randomly distributed but clustered within specific hotspots (<xref ref-type="fig" rid="figure3">Figure 3</xref>). Language-specific heatmaps are provided in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>, whereas <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> provides a sentence-number mapping guide for the 43 source segments. Because the EQ-5D-5L source instrument is proprietary, this mapping is presented as annotated screenshots from the sample UK English digital questionnaire in the EuroQol user guide rather than as a reprinted text table.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Sentence-level COMET heatmap across 5 target languages and 4 AI translation services in this methodological benchmarking study of EQ-5D-5L translation. Rows correspond to the 43 source segments and columns correspond to the AI services within each language panel. Darker green indicates higher semantic similarity to the official, linguistically validated human translation, whereas lighter colors indicate lower scores and localized translation hotspots. The figure illustrates that most segments scored highly across services, but lower-scoring deviations clustered in specific source segments rather than being randomly distributed across the instrument. See <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> for the sentence-number mapping guide. COMET: cross-lingual optimized metric for evaluation of translation. The target languages are represented in the figure as follows: DA: Danish; NL: Dutch; FR: French; DE: German; and ES: Spanish. EQ-5D-5L: EuroQol 5-dimension 5-level.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e78485_fig03.png"/></fig><p>Some of these hotspots corresponded to short digital interface elements rather than clinical questionnaire constructs. In particular, the final source segments included navigation and interface strings such as &#x201C;Previous,&#x201D; &#x201C;Next,&#x201D; and an error message. These elements are context-dependent and may be difficult to translate optimally when evaluated as isolated sentence-level strings outside their software environment. Therefore, low scores for these segments should be interpreted separately from deviations involving clinical concepts, domain headers, or response options.</p><p>BLEU identified the largest hotspot clusters in Spanish and Dutch, indicating where review effort may need to be concentrated for specific language-service combinations. However, because hotspot counts alone do not establish overall adequacy, overall translation quality was interpreted based on the full score distributions and semantic metric patterns, and heatmaps were used specifically to localize low-scoring deviations.</p><p>To illustrate the clinical implications of these findings, we conducted a qualitative review of both high-scoring and low-scoring sentences (<xref ref-type="table" rid="table2">Table 2</xref>). For simple declarative sentences, the AI models frequently achieved perfect alignment with the human reference translations, producing identical, verbatim matches across all 4 services (eg, the German translation for pain severity). This confirms that, for standard grammatical structures and unambiguous phrasing, AI performance was highly reliable and consistent.</p><p>However, for section headers, abstract health concepts, and short context-dependent interface strings, some differences from the gold standard emerged. As detailed in <xref ref-type="table" rid="table2">Table 2</xref>, AI services sometimes defaulted to literal or highly medicalized terminology rather than to the intended clinical concept. This pattern was observed across languages; for instance, in Danish, models frequently selected terms that were technically correct but contextually distinct (eg, translating &#x201C;Mobility&#x201D; as the sociological concept <italic>Mobilitet</italic> rather than the physical capacity <italic>Bev&#x00E6;gelighed</italic> used in the validated version). These examples illustrate a possible limitation of current AI models in distinguishing literal translation from conceptual equivalence. The practical relevance of isolated header differences could not be assessed in the present study. Further equivalence or cognitive-debriefing research will be needed to determine which discrepancies are acceptable and which require the involvement of expert human linguists. Therefore, <xref ref-type="table" rid="table2">Table 2</xref> presents representative examples of both high-scoring exact matches and low-scoring conceptual deviations, while the heatmap and sentence-number mapping identify additional hotspots, including interface-related strings.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Representative high-scoring and low-scoring examples from this methodological benchmarking study of AI-powered EuroQol 5-dimension 5-level (EQ-5D-5L) translation<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Language (code)</td><td align="left" valign="bottom">Source text</td><td align="left" valign="bottom">Gold standard translation</td><td align="left" valign="bottom">AI translation (service)</td><td align="left" valign="bottom">Metric score</td><td align="left" valign="bottom">Error type</td><td align="left" valign="bottom">Qualitative error description</td></tr></thead><tbody><tr><td align="left" valign="top">German (DE)</td><td align="left" valign="top">I have severe pain or discomfort</td><td align="left" valign="top">Ich habe starke Schmerzen oder Beschwerden</td><td align="left" valign="top">Ich habe starke Schmerzen oder Beschwerden (Google)</td><td align="left" valign="top">BLEU<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>: 1.0 METEOR<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup>: 1.0 COMET<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup>: 0.983 BLEURT<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup>: 0.964</td><td align="left" valign="top">None</td><td align="left" valign="top">All 4 services produced an identical match to the human reference, demonstrating high reliability for simple declarative sentences.</td></tr><tr><td align="left" valign="top">Danish (DA)</td><td align="left" valign="top">The best health you can imagine</td><td align="left" valign="top">Det bedste helbred, du kan forestille dig</td><td align="left" valign="top">Det bedste helbred, du kan forestille dig (GPT-4.1)</td><td align="left" valign="top">BLEU: 1.0 METEOR: 1.0 COMET: 0.975 BLEURT: 1.0</td><td align="left" valign="top">None</td><td align="left" valign="top">All 4 services produced an identical match to the human reference, demonstrating high reliability for simple declarative sentences.</td></tr><tr><td align="left" valign="top">French (FR)</td><td align="left" valign="top">This scale is numbered from 0 to 100.</td><td align="left" valign="top">Cette &#x00E9;chelle est num&#x00E9;rot&#x00E9;e de 0 &#x00E0; 100</td><td align="left" valign="top">Cette &#x00E9;chelle est num&#x00E9;rot&#x00E9;e de 0 &#x00E0; 100 (Amazon)</td><td align="left" valign="top">BLEU: 1.0 METEOR: 1.0 COMET: 0.984 BLEURT: 0.968</td><td align="left" valign="top">None</td><td align="left" valign="top">All 4 services produced an identical match to the human reference, demonstrating high reliability for simple declarative sentences.</td></tr><tr><td align="left" valign="top">French (FR)</td><td align="left" valign="top">Self-care</td><td align="left" valign="top">AUTONOMIE DE LA PERSONNE</td><td align="left" valign="top">SOINS AUTO-ADMINISTR&#x00C9;S (Google)</td><td align="left" valign="top">BLEU: 0.0 METEOR: 0.0 COMET: 0.398 BLEURT: 0.198</td><td align="left" valign="top">Conceptual</td><td align="left" valign="top">Literal translation implies &#x201C;medical treatment&#x201D; (administering care) rather than the clinical concept of &#x201C;personal autonomy&#x201D; (washing or dressing).</td></tr><tr><td align="left" valign="top">German (DE)</td><td align="left" valign="top">Pain or discomfort</td><td align="left" valign="top">SCHMERZEN / K&#x00D6;RPERLICHE BESCHWERDEN</td><td align="left" valign="top">SCHMERZ / UNWOHLSEIN (Amazon)</td><td align="left" valign="top">BLEU: 0.0 METEOR: 0.0 COMET: 0.623 BLEURT: 0.445</td><td align="left" valign="top">Semantic</td><td align="left" valign="top">The AI translation &#x201C;Unwohlsein&#x201D; (feeling unwell or malaise) is a general subjective state, whereas the gold standard Beschwerden (complaints or discomfort) is a broader term encompassing specific physical pain or functional issues.</td></tr><tr><td align="left" valign="top">Dutch (NL)</td><td align="left" valign="top">Anxiety or depression</td><td align="left" valign="top">ANGST / SOMBERHEID</td><td align="left" valign="top">ANGST / DEPRESSIE (DeepL)</td><td align="left" valign="top">BLEU: 0.0 METEOR: 0.623 COMET: 0.419 BLEURT: 0.580</td><td align="left" valign="top">Intensity</td><td align="left" valign="top">The AI translation &#x201C;Depressie&#x201D; refers to a clinical psychiatric disorder (pathology), whereas the Gold Standard &#x201C;Somberheid&#x201D; captures the subjective feeling of gloom or sadness (symptom) required for general health reporting.</td></tr><tr><td align="left" valign="top">Danish (DA)</td><td align="left" valign="top">Mobility</td><td align="left" valign="top">BEV&#x00C6;GELIGHED</td><td align="left" valign="top">MOBILITET (GPT-4.1)</td><td align="left" valign="top">BLEU: 0.0 METEOR: 0.0 COMET: 0.341 BLEURT: 0.271</td><td align="left" valign="top">Domain</td><td align="left" valign="top">The AI translation &#x201C;Mobilitet&#x201D; is used for sociology or transport (systemic movement), whereas the Gold Standard &#x201C;Bev&#x00E6;gelighed&#x201D; refers to biological/physical capacity (range of motion).</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Examples are drawn from sentence-level comparisons between AI-generated translations and official linguistically validated human translations (gold standard) across 5 target languages and illustrate both exact matches and conceptually important deviations identified during hotspot analysis. Conceptual indicates incorrect rendering of the intended clinical construct; semantic indicates a meaning shift relative to the gold standard; intensity indicates altered severity or strength; and domain indicates a shift into an inappropriate usage context or domain.</p></fn><fn id="table2fn2"><p><sup>b</sup>BLEU: bilingual evaluation understudy.</p></fn><fn id="table2fn3"><p><sup>c</sup>METEOR: metric for evaluation of translation with explicit ordering.</p></fn><fn id="table2fn4"><p><sup>d</sup>COMET: cross-lingual optimized metric for evaluation of translation.</p></fn><fn id="table2fn5"><p><sup>e</sup>BLEURT: bilingual evaluation understudy with representations from transformers.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study reveals that although AI translation services are not statistically identical, most between-service differences were stylistic rather than semantic. We identified 11 statistically significant pairwise differences (adjusted <italic>P</italic>&#x003C;.05, ranging from &#x003C;.001 to .049), of which 91% (10 of 11) were detected by surface-overlap metrics (BLEU/METEOR), whereas only 1 was detected by a semantic metric (BLEURT). Supported by the generally small omnibus effect sizes (median Kendall <italic>W</italic>=0.08, IQR 0.07-0.10), this pattern suggests that although services differed in lexical choice, they produced translations with highly comparable semantic meaning relative to the validated human translations. Taken together, the semantic score distributions and the localized nature of the low-scoring hotspots suggest that current AI services can generate translations for this instrument in the evaluated languages. However, this interpretation depends on the combined descriptive, semantic, and hotspot analyses rather than on significance testing alone. Some differences occurred in parts of the questionnaire outside the scored response options; however, their practical relevance for participant interpretation could not be assessed in the present study. Further research will be needed to determine which PROM elements (ie, instructions and answer options) should be focused on when evaluating translation quality.</p></sec><sec id="s4-2"><title>Defining the Human-in-the-Loop</title><p>The finding of statistical comparability of average scores must be interpreted with caution, as aggregate metrics can mask critical, low-incidence discrepancies. Our qualitative outlier analysis revealed that deviations from the gold standard were evenly distributed across all services (85&#x2010;92 outliers per service). The lowest-scoring deviations were primarily observed in domain headers, and the 4 representative low-scoring examples in <xref ref-type="table" rid="table2">Table 2</xref> came from different AI services. These findings suggest that human oversight may be required across services, although further research is needed to determine the level of review required to ensure acceptable translation quality.</p><p>The potential clinical significance of these deviations was evaluated through a manual review of systemic inconsistencies against the official, linguistically validated translations. As detailed in <xref ref-type="table" rid="table2">Table 2</xref>, semantic inconsistencies occurred primarily in abstract concepts and domain headers rather than in the actual response options themselves. For instance, the domain header concept &#x201C;Self-care&#x201D; was frequently translated literally (eg, Soins auto-administr&#x00E9;s or &#x201C;self-administered medical care&#x201D;) rather than conceptually (referring to personal autonomy in washing or dressing). Although the subsequent item text (&#x201C;I have no problems washing or dressing&#x201D;) was often translated correctly, the practical relevance of isolated header differences could not be assessed in the present study. This interpretation is consistent with electronic clinical outcome assessment migration literature, which distinguishes core respondent-facing content, such as instructions, item wording, and response options, from other contextual or presentation elements when considering the level of equivalence evidence required [<xref ref-type="bibr" rid="ref2">2</xref>]. A follow-up study is planned to quantify the reduction in effort that AI can offer compared with fully manual translation.</p></sec><sec id="s4-3"><title>Limitations</title><p>We acknowledge several limitations. First, cultural adaptation was assessed only indirectly, as no patient cognitive debriefing was performed. Second, the study used sentence-level translation and evaluation to support aligned benchmarking against the gold standard, which may have constrained systems that benefit from broader document context and may not have fully captured dependence among related instrument segments. Third, the findings are limited to high-resource European languages and should not be generalized to lower-resource or structurally distinct languages without further validation. Fourth, the primary semantic metrics were reference-based and may therefore have penalized acceptable alternative phrasings that differed from the validated reference wording. Fifth, the service configuration was not fully symmetrical because GPT-4.1 was used with a structured medical translation prompt, whereas Google Translate, Amazon Translate, and DeepL were evaluated in their standard API mode. Sixth, because the EQ-5D-5L and its validated translations are widely used, some systems may have been exposed to similar wording during training or optimization. Finally, the qualitative error categories were interpretive and descriptive rather than based on a formally validated annotation framework, and no blinded, independent, multirater review or formal interrater reliability testing was performed.</p></sec><sec id="s4-4"><title>Future Research Directions</title><p>Future work will extend this benchmarking approach in several directions. Automated metric results should first be compared directly with expert human judgments to determine how well these scores reflect professional linguistic review of PROM translations. Studies should also quantify the operational value of human-in-the-loop workflows by comparing fully manual translation with AI-assisted translation followed by expert editing. Broader evaluation designs should examine whether performance changes when related questionnaire elements are analyzed at the domain level rather than strictly at the sentence level, and whether these findings generalize to structurally different or lower-resource languages. Finally, future studies should explore hybrid and reference-free evaluation strategies, including quality estimation methods, and assess how such approaches could be integrated into clinical trial translation workflows more effectively.</p></sec><sec id="s4-5"><title>Conclusions</title><p>This benchmarking study showed that leading AI translation services were not functionally interchangeable across metric-language combinations. Although statistically significant differences were identified, these were more often detected by surface-overlap metrics than by semantic metrics. These findings support the use of AI to generate PROM translations for this instrument and the evaluated high-resource European languages. Expert human linguist review may need to be retained to confirm conceptual equivalence and fitness for clinical use.</p></sec></sec></body><back><ack><p>The authors gratefully acknowledge the EuroQol Research Foundation for providing access to the EQ-5D-5L (EuroQol 5-dimension 5-level) questionnaire and its validated translations for the purposes of this academic research. The authors also wish to express their gratitude to Prof Tom&#x00E1;s Ward for his supervision throughout this research. ChatGPT (OpenAI), using multiple model versions, was used for language editing, proofreading, grammar correction, and readability improvement during manuscript preparation. It was not used to generate or analyze study data or to make scientific, methodological, or interpretive decisions. All scientific content, analytic decisions, interpretations, and the final manuscript were reviewed and approved by the authors.</p></ack><notes><sec><title>Funding</title><p>The authors declared no financial support was received for this work.</p></sec><sec><title>Data Availability</title><p>The EQ-5D-5L (EuroQol 5-dimension 5-level) questionnaire and its validated translations are proprietary materials of the EuroQol Research Foundation and were used for methodological benchmarking rather than human-participant research. Accordingly, these specific datasets cannot be publicly shared or made available by the authors. However, the methodology used and the aggregated statistical results are fully described within this manuscript to ensure transparency and reproducibility of the research process.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: HV, WM</p><p>Data curation: HV</p><p>Formal analysis: HV</p><p>Investigation: HV</p><p>Methodology: HV, WM</p><p>Software: HV</p><p>Supervision: TW</p><p>Validation: WM</p><p>Visualization: HV</p><p>Writing &#x2013; original draft: HV</p><p>Writing &#x2013; review &#x0026; editing: HV, TW, WM</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">API</term><def><p>application programming interface</p></def></def-item><def-item><term id="abb2">AWS</term><def><p>Amazon Web Services</p></def></def-item><def-item><term id="abb3">BLEU</term><def><p>bilingual evaluation understudy</p></def></def-item><def-item><term id="abb4">BLEURT</term><def><p>bilingual evaluation understudy with representations from transformers</p></def></def-item><def-item><term id="abb5">COMET</term><def><p>cross-lingual optimized metric for evaluation of translation</p></def></def-item><def-item><term id="abb6">EQ-5D-5L</term><def><p>EuroQol 5-dimension 5-level</p></def></def-item><def-item><term id="abb7">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb8">METEOR</term><def><p>metric for evaluation of translation with explicit ordering</p></def></def-item><def-item><term id="abb9">NMT</term><def><p>neural machine translation</p></def></def-item><def-item><term id="abb10">PROM</term><def><p>patient-reported outcome measure</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Acquadro</surname><given-names>C</given-names> </name><name name-style="western"><surname>Conway</surname><given-names>K</given-names> </name><name name-style="western"><surname>Girourdet</surname><given-names>C</given-names> </name><name name-style="western"><surname>Mear</surname><given-names>I</given-names> </name></person-group><source>Linguistic Validation Manual for Patient-Reported Outcomes (PRO) Instruments</source><year>2004</year><access-date>2026-07-22</access-date><publisher-name>Mapi Research Trust</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://books.google.co.in/books?id=ldj-MgEACAAJ">https://books.google.co.in/books?id=ldj-MgEACAAJ</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Byrom</surname><given-names>B</given-names> </name><name name-style="western"><surname>Gwaltney</surname><given-names>C</given-names> </name><name name-style="western"><surname>Slagle</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gnanasakthy</surname><given-names>A</given-names> </name><name name-style="western"><surname>Muehlhausen</surname><given-names>W</given-names> </name></person-group><article-title>Measurement equivalence of patient-reported outcome measures migrated to electronic formats: a review of evidence and recommendations for clinical trials and bring your own device</article-title><source>Ther Innov Regul Sci</source><year>2019</year><month>07</month><volume>53</volume><issue>4</issue><fpage>426</fpage><lpage>430</lpage><pub-id pub-id-type="doi">10.1177/2168479018793369</pub-id><pub-id pub-id-type="medline">30157687</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eremenco</surname><given-names>SL</given-names> </name><name name-style="western"><surname>Cella</surname><given-names>D</given-names> </name><name name-style="western"><surname>Arnold</surname><given-names>BJ</given-names> </name></person-group><article-title>A comprehensive method for the translation and cross-cultural validation of health status questionnaires</article-title><source>Eval Health Prof</source><year>2005</year><month>06</month><volume>28</volume><issue>2</issue><fpage>212</fpage><lpage>232</lpage><pub-id pub-id-type="doi">10.1177/0163278705275342</pub-id><pub-id pub-id-type="medline">15851774</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wild</surname><given-names>D</given-names> </name><name name-style="western"><surname>Grove</surname><given-names>A</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Principles of good practice for the translation and cultural adaptation process for patient-reported outcomes (PRO) measures: report of the ISPOR Task Force for Translation and Cultural Adaptation</article-title><source>Value Health</source><year>2005</year><volume>8</volume><issue>2</issue><fpage>94</fpage><lpage>104</lpage><pub-id pub-id-type="doi">10.1111/j.1524-4733.2005.04054.x</pub-id><pub-id pub-id-type="medline">15804318</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Patrick</surname><given-names>DL</given-names> </name><name name-style="western"><surname>Burke</surname><given-names>LB</given-names> </name><name name-style="western"><surname>Gwaltney</surname><given-names>CJ</given-names> </name><etal/></person-group><article-title>Content validity&#x2014;establishing and reporting the evidence in newly developed patient-reported outcomes (PRO) instruments for medical product evaluation: ISPOR PRO Good Research Practices Task Force report: part 2&#x2014;assessing respondent understanding</article-title><source>Value Health</source><year>2011</year><month>12</month><volume>14</volume><issue>8</issue><fpage>978</fpage><lpage>988</lpage><pub-id pub-id-type="doi">10.1016/j.jval.2011.06.013</pub-id><pub-id pub-id-type="medline">22152166</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Harkness</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Villar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Edwards</surname><given-names>B</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Harkness</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Braun</surname><given-names>M</given-names> </name><name name-style="western"><surname>Edwards</surname><given-names>B</given-names> </name><name name-style="western"><surname>Johnson</surname><given-names>TP</given-names> </name><name name-style="western"><surname>Lyberg</surname><given-names>L</given-names> </name><name name-style="western"><surname>Mohler</surname><given-names>PP</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>TW</given-names> </name></person-group><article-title>Translation, adaptation, and design</article-title><source>Survey Methods in Multinational, Multiregional, and Multicultural Contexts</source><year>2010</year><publisher-name>John Wiley &#x0026; Sons</publisher-name><fpage>115</fpage><lpage>140</lpage><pub-id pub-id-type="doi">10.1002/9780470609927</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Foote</surname><given-names>HP</given-names> </name><name name-style="western"><surname>Hong</surname><given-names>C</given-names> </name><name name-style="western"><surname>Anwar</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Embracing generative artificial intelligence in clinical research and beyond: opportunities, challenges, and solutions</article-title><source>JACC Adv</source><year>2025</year><month>03</month><volume>4</volume><issue>3</issue><fpage>101593</fpage><pub-id pub-id-type="doi">10.1016/j.jacadv.2025.101593</pub-id><pub-id pub-id-type="medline">39923329</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>van Kolfschooten</surname><given-names>H</given-names> </name><name name-style="western"><surname>Goosen</surname><given-names>S</given-names> </name><name name-style="western"><surname>van Oirschot</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schouten</surname><given-names>B</given-names> </name><name name-style="western"><surname>Vajda</surname><given-names>I</given-names> </name><name name-style="western"><surname>Willems</surname><given-names>L</given-names> </name></person-group><article-title>Legal, ethical, and policy challenges of artificial intelligence translation tools in healthcare</article-title><source>Discov Public Health</source><year>2025</year><volume>22</volume><issue>1</issue><fpage>904</fpage><pub-id pub-id-type="doi">10.1186/s12982-025-01277-z</pub-id><pub-id pub-id-type="medline">41487681</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Johnson</surname><given-names>M</given-names> </name><name name-style="western"><surname>Schuster</surname><given-names>M</given-names> </name><name name-style="western"><surname>Le</surname><given-names>QV</given-names> </name><etal/></person-group><article-title>Google's multilingual neural machine translation system: enabling zero-shot translation</article-title><source>Trans Assoc Comput Linguist</source><year>2017</year><month>12</month><volume>5</volume><fpage>339</fpage><lpage>351</lpage><pub-id pub-id-type="doi">10.1162/tacl_a_00065</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Schuster</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Google&#x2019;s neural machine translation system: bridging the gap between human and machine translation</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 26, 2016</comment><pub-id pub-id-type="doi">10.48550/arXiv.1609.08144</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Solomou</surname><given-names>T</given-names> </name><name name-style="western"><surname>Mappouras</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kyriacou</surname><given-names>E</given-names> </name><etal/></person-group><article-title>Bridging language barriers in healthcare: a patient-centric mobile app for multilingual health record access and sharing</article-title><source>Front Digit Health</source><year>2025</year><volume>7</volume><fpage>1542485</fpage><pub-id pub-id-type="doi">10.3389/fdgth.2025.1542485</pub-id><pub-id pub-id-type="medline">40041125</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Nguyen</surname><given-names>HH</given-names> </name><name name-style="western"><surname>Mahajan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Yadav</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Prompting with phonemes: enhancing llms&#x2019; multilinguality for non-latin script languages</article-title><conf-name>2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics</conf-name><conf-date>Apr 29 to May 4, 2025</conf-date><conf-loc>Albuquerque, New Mexico</conf-loc><fpage>11975</fpage><lpage>11994</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.naacl-long.599</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lu</surname><given-names>SC</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Kaur</surname><given-names>M</given-names> </name><name name-style="western"><surname>Edelen</surname><given-names>MO</given-names> </name><name name-style="western"><surname>Pusic</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gibbons</surname><given-names>C</given-names> </name></person-group><article-title>Can machine translation match human expertise? Quantifying the performance of large language models in the translation of patient-reported outcome measures (PROMs)</article-title><source>J Patient Rep Outcomes</source><year>2025</year><month>07</month><day>25</day><volume>9</volume><issue>1</issue><fpage>94</fpage><pub-id pub-id-type="doi">10.1186/s41687-025-00926-w</pub-id><pub-id pub-id-type="medline">40711496</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kwong</surname><given-names>JCC</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>SCY</given-names> </name><name name-style="western"><surname>Nickel</surname><given-names>GC</given-names> </name><name name-style="western"><surname>Cacciamani</surname><given-names>GE</given-names> </name><name name-style="western"><surname>Kvedar</surname><given-names>JC</given-names> </name></person-group><article-title>The long but necessary road to responsible use of large language models in healthcare research</article-title><source>NPJ Digit Med</source><year>2024</year><month>07</month><day>4</day><volume>7</volume><issue>1</issue><fpage>177</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01180-y</pub-id><pub-id pub-id-type="medline">38965411</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="web"><article-title>Quickstart</article-title><source>DeepL API documentation</source><access-date>2025-07-28</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://developers.deepl.com/docs/getting-started/quickstart">https://developers.deepl.com/docs/getting-started/quickstart</ext-link></comment></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="web"><article-title>Amazon Translate documentation</article-title><source>Amazon Web Services</source><access-date>2025-07-28</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://docs.aws.amazon.com/translate/">https://docs.aws.amazon.com/translate/</ext-link></comment></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="web"><article-title>Models</article-title><source>OpenAI Developers</source><access-date>2025-07-28</access-date><publisher-name>OpenAI</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://platform.openai.com/docs/models">https://platform.openai.com/docs/models</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ataman</surname><given-names>D</given-names> </name><name name-style="western"><surname>Birch</surname><given-names>A</given-names> </name><name name-style="western"><surname>Habash</surname><given-names>N</given-names> </name><name name-style="western"><surname>Federico</surname><given-names>M</given-names> </name><name name-style="western"><surname>Koehn</surname><given-names>P</given-names> </name><name name-style="western"><surname>Cho</surname><given-names>K</given-names> </name></person-group><article-title>Machine translation in the era of large language models: a survey of historical and emerging problems</article-title><source>Information</source><year>2025</year><volume>16</volume><issue>9</issue><fpage>723</fpage><pub-id pub-id-type="doi">10.3390/info16090723</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Herdman</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gudex</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lloyd</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Development and preliminary testing of the new five-level version of EQ-5D (EQ-5D-5L)</article-title><source>Qual Life Res</source><year>2011</year><month>12</month><volume>20</volume><issue>10</issue><fpage>1727</fpage><lpage>1736</lpage><pub-id pub-id-type="doi">10.1007/s11136-011-9903-x</pub-id><pub-id pub-id-type="medline">21479777</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="report"><article-title>EQ-5D-5L User Guide Version 3.0</article-title><year>2019</year><access-date>2025-03-02</access-date><publisher-name>EuroQol Research Foundation</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://euroqol.org/wp-content/uploads/2023/11/EQ-5D-5LUserguide-23-07.pdf">https://euroqol.org/wp-content/uploads/2023/11/EQ-5D-5LUserguide-23-07.pdf</ext-link></comment></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Janssen</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Pickard</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Golicki</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Measurement properties of the EQ-5D-5L compared to the EQ-5D-3L across eight patient groups: a multi-country study</article-title><source>Qual Life Res</source><year>2013</year><month>09</month><volume>22</volume><issue>7</issue><fpage>1717</fpage><lpage>1727</lpage><pub-id pub-id-type="doi">10.1007/s11136-012-0322-4</pub-id><pub-id pub-id-type="medline">23184421</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Papineni</surname><given-names>K</given-names> </name><name name-style="western"><surname>Roukos</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ward</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>WJ</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Isabelle</surname><given-names>P</given-names> </name><name name-style="western"><surname>Charniak</surname><given-names>E</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>D</given-names> </name></person-group><article-title>BLEU: a method for automatic evaluation of machine translation</article-title><source>Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics</source><year>2002</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>311</fpage><lpage>318</lpage><pub-id pub-id-type="doi">10.3115/1073083.1073135</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Banerjee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lavie</surname><given-names>A</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Goldstein</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lavie</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>CY</given-names> </name><name name-style="western"><surname>Voss</surname><given-names>C</given-names> </name></person-group><article-title>METEOR: an automatic metric for MT evaluation with improved correlation with human judgments</article-title><source>Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and/or Summarization</source><year>2005</year><access-date>2025-04-28</access-date><publisher-name>Association for Computational Linguistics</publisher-name><fpage>65</fpage><lpage>72</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/W05-0909.pdf">https://aclanthology.org/W05-0909.pdf</ext-link></comment></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Rei</surname><given-names>R</given-names> </name><name name-style="western"><surname>Stewart</surname><given-names>C</given-names> </name><name name-style="western"><surname>Farinha</surname><given-names>AC</given-names> </name><name name-style="western"><surname>Lavie</surname><given-names>A</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Webber</surname><given-names>B</given-names> </name><name name-style="western"><surname>Cohn</surname><given-names>T</given-names> </name><name name-style="western"><surname>He</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name></person-group><article-title>COMET: a neural framework for MT evaluation</article-title><source>Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing</source><year>2020</year><access-date>2026-07-28</access-date><publisher-name>Association for Computational Linguistics</publisher-name><fpage>2685</fpage><lpage>2702</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2020.emnlp-main.213.pdf">https://aclanthology.org/2020.emnlp-main.213.pdf</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Sellam</surname><given-names>T</given-names> </name><name name-style="western"><surname>Das</surname><given-names>D</given-names> </name><name name-style="western"><surname>Parikh</surname><given-names>A</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Jurafsky</surname><given-names>D</given-names> </name><name name-style="western"><surname>Chai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schluter</surname><given-names>N</given-names> </name><name name-style="western"><surname>Tetreault</surname><given-names>J</given-names> </name></person-group><article-title>BLEURT: learning robust metrics for text generation</article-title><source>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics</source><year>2020</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>7881</fpage><lpage>7892</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.704</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guerreiro</surname><given-names>NM</given-names> </name><name name-style="western"><surname>Rei</surname><given-names>R</given-names> </name><name name-style="western"><surname>van Stigt</surname><given-names>D</given-names> </name><name name-style="western"><surname>Coheur</surname><given-names>L</given-names> </name><name name-style="western"><surname>Colombo</surname><given-names>P</given-names> </name><name name-style="western"><surname>Martins</surname><given-names>AFT</given-names> </name></person-group><article-title>xCOMET: transparent machine translation evaluation through fine-grained error detection</article-title><source>Trans Assoc Comput Linguist</source><year>2024</year><volume>12</volume><fpage>979</fpage><lpage>995</lpage><pub-id pub-id-type="doi">10.1162/tacl_a_00683</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1 </label><p>System prompt and reproducibility parameters for the GPT-4.1 model.</p><media xlink:href="ai_v5i1e78485_app1.docx" xlink:title="DOCX File, 20 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Annotated sentence-number mapping guide for the 43 EQ-5D-5L source segments based on the sample UK English digital questionnaire from the official EuroQol user guide.</p><media xlink:href="ai_v5i1e78485_app2.docx" xlink:title="DOCX File, 1546 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Supplementary violin plots, heatmaps, and threshold-based descriptive summaries of sentence-level metric scores for each target language.</p><media xlink:href="ai_v5i1e78485_app3.docx" xlink:title="DOCX File, 1403 KB"/></supplementary-material></app-group></back></article>