<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR AI</journal-id><journal-id journal-id-type="publisher-id">ai</journal-id><journal-id journal-id-type="index">41</journal-id><journal-title>JMIR AI</journal-title><abbrev-journal-title>JMIR AI</abbrev-journal-title><issn pub-type="epub">2817-1705</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v5i1e103321</article-id><article-id pub-id-type="doi">10.2196/103321</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Multilingual Disparities in Large Language Model&#x2013;Based Symptom Detection for Global Disease Surveillance: Evaluation Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Jannah</surname><given-names>Sa'idah Zahrotul</given-names></name><degrees>MStat</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Nishiyama</surname><given-names>Tomohiro</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Peng</surname><given-names>Shaowen</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wakamiya</surname><given-names>Shoko</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Aramaki</surname><given-names>Eiji</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Nara Institute of Science and Technology</institution><addr-line>8916-5 Takayama-cho</addr-line><addr-line>Ikoma</addr-line><addr-line>Nara</addr-line><country>Japan</country></aff><aff id="aff2"><institution>Universitas Airlangga</institution><addr-line>Surabaya</addr-line><addr-line>Jawa Timur</addr-line><country>Indonesia</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Steenstra</surname><given-names>Ivan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Ozsoy</surname><given-names>Makbule Gulcin</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Vassileva</surname><given-names>Sylvia</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Eiji Aramaki, PhD, Nara Institute of Science and Technology, 8916-5 Takayama-cho, Ikoma, Nara, 630-0192, Japan, 81 743725111; <email>aramaki@is.naist.jp</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>9</day><month>10</month><year>2026</year></pub-date><volume>5</volume><elocation-id>e103321</elocation-id><history><date date-type="received"><day>02</day><month>06</month><year>2026</year></date><date date-type="rev-recd"><day>28</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>31</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Sa'idah Zahrotul Jannah, Tomohiro Nishiyama, Shaowen Peng, Shoko Wakamiya, Eiji Aramaki. Originally published in JMIR AI (<ext-link ext-link-type="uri" xlink:href="https://ai.jmir.org">https://ai.jmir.org</ext-link>), 9.10.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR AI, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://www.ai.jmir.org/">https://www.ai.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://ai.jmir.org/2026/1/e103321"/><abstract><sec><title>Background</title><p>Symptom detection is essential in global disease surveillance to detect potential outbreaks, as symptoms are the first observable signs of infection. To reflect real-time ground truth conditions during pandemics, social media has emerged as a valuable data source. Moreover, effective digital disease surveillance systems must operate across diverse linguistic settings, and large language models (LLMs) have been shown to perform inconsistently across languages, tending to have lower performance in low-resource languages. While multilingual approaches have been explored in various health-related natural language processing tasks, a critical gap remains in understanding whether LLM-based symptom detection can perform consistently across languages for global disease surveillance. Southeast Asia demonstrates this challenge, combining diverse languages and the potential for emerging infectious disease outbreaks, making it a case for evaluating how multilingual performance disparities manifest in symptom detection.</p></sec><sec><title>Objective</title><p>This study aims to evaluate multilingual disparities in symptom detection using a LLM, as well as the associated error mechanisms across languages and symptom types, to better understand their implications for global disease surveillance.</p></sec><sec sec-type="methods"><title>Methods</title><p>This study uses the MedWeb dataset, a multilingual pseudo&#x2013;social media text dataset with multiple symptom labels. The data consist of 12 languages, covering diverse regions and language resource classifications. The symptoms included in this study are fever, headache, runny nose, cough, diarrhea, hay fever, influenza, and cold. We used GPT-5 as the symptom detection system, representing a strong model for health-related tasks. The results are evaluated using the macro precision, recall, and <italic>F</italic><sub>1</sub>-score. An error analysis was conducted to identify the error mechanisms underlying incorrect symptom predictions across languages and to calculate the impact of each error category.</p></sec><sec sec-type="results"><title>Results</title><p>Our findings show that performance varied across languages, language resource groups, and symptom types. Southeast Asian (SEA) languages generally achieved lower scores than non-SEA languages, with Japanese obtaining the highest <italic>F</italic><sub>1</sub>-score (0.812) and Lao the lowest (0.716). High-resource languages achieved the most consistent performance, while low-resource languages obtained the lowest overall scores. At the symptom level, diarrhea, headache, cough, and fever showed stable detection across languages, while hay fever, runny nose, influenza, and cold exhibited greater variability. Hay fever showed the widest variability, forming 2 distinct performance clusters aligned with language resource and region classifications. The error analysis revealed 4 misclassification patterns: explicitly mentioned symptoms, cross-lingual variation, symptom overgeneralization, and context misinterpretation. Cross-lingual variation was the most frequent error category, while errors related to explicitly mentioned symptoms showed the largest potential impact on model performance.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Due to performance disparities, achieving more reliable and equitable LLM-based symptom detection from social media text for global disease surveillance would benefit from broader representation in training data for low-resource languages, improved cultural and linguistic sensitivity, and stronger contextual understanding of symptom-related expressions.</p></sec></abstract><kwd-group><kwd>symptom detection</kwd><kwd>global disease surveillance</kwd><kwd>multilingual disparities</kwd><kwd>large language models</kwd><kwd>low-resource language</kwd><kwd>social media</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Global disease surveillance plays an important role in public health by detecting potential disease outbreaks. Outbreaks can be identified through symptoms, which are often the first observable signs in the human body after infection [<xref ref-type="bibr" rid="ref1">1</xref>]. Information shared by social media users during an outbreak can reflect ground truth conditions [<xref ref-type="bibr" rid="ref2">2</xref>]. This makes social media text a potential data source for detecting symptoms in public health surveillance [<xref ref-type="bibr" rid="ref3">3</xref>-<xref ref-type="bibr" rid="ref5">5</xref>].</p><p>Social media data used for digital disease surveillance operate across diverse linguistic settings, requiring reliable systems capable of processing multilingual data. This is particularly important for symptom detection, as such systems require not only linguistic understanding but also the ability to interpret symptom expressions across languages. However, the availability of language resources varies across languages. In natural language processing (NLP), languages can be classified along a spectrum ranging from high resource to low resource, depending on factors such as the availability of labeled and unlabeled data [<xref ref-type="bibr" rid="ref6">6</xref>]. Previous studies have reported that large language models (LLMs) tend to perform better on high-resource languages [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>While prior studies have explored multilingual approaches in health-related NLP tasks, several critical gaps remain. Previous work has evaluated language models&#x2019; performance across multiple languages in medical question-answering from clinical text [<xref ref-type="bibr" rid="ref9">9</xref>], classification of epidemiological characteristics of infectious disease outbreaks [<xref ref-type="bibr" rid="ref10">10</xref>], symptom entity recognition from clinical text [<xref ref-type="bibr" rid="ref11">11</xref>], and adverse drug reaction detection [<xref ref-type="bibr" rid="ref12">12</xref>]. A previous study evaluated LLM-based symptom detection across 7 languages, showing that, on average, European languages outperformed Asian languages and that influenza was significantly overpredicted across all languages [<xref ref-type="bibr" rid="ref13">13</xref>]. However, these studies did not evaluate how detection performance varies across a broader range of language resource levels, particularly for low-resource languages in the context of disease surveillance. This leaves a gap in understanding whether LLM-based symptom detection systems can perform consistently across diverse language resource levels for global disease surveillance. This gap is particularly critical in regions characterized by linguistic diversity and limited language resources, where the population is also at risk of becoming the epicenter of disease outbreaks.</p><p>Southeast Asia presents this challenge, as it comprises diverse midresource to low-resource languages and exhibits relatively lower model performance in these languages [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. Moreover, the region has frequently been identified as a source of emerging infectious diseases, highlighting the need for targeted intervention strategies to mitigate outbreak risks [<xref ref-type="bibr" rid="ref16">16</xref>-<xref ref-type="bibr" rid="ref19">19</xref>]. This makes Southeast Asia a particularly relevant region for examining symptom detection across diverse regional languages and comparing model performance with higher-resource languages.</p><p>Therefore, to address this gap, we specifically evaluate an LLM on a multilingual social media dataset covering 12 languages with diverse resource classifications for multilabel symptom detection. <xref ref-type="fig" rid="figure1">Figure 1</xref> provides an overview of this study: the input is a multilingual, social media&#x2013;based dataset covering diverse languages and regions, which is then processed by an LLM. The output consists of symptoms detected by the system. We evaluated the results from both language and symptom perspectives, then analyzed the misclassification patterns. Overall, this study design addresses the research objective of evaluating multilingual disparities in symptom detection using an LLM across languages and symptom types to better understand their implications for global disease surveillance. It contributes to identifying multilingual performance disparities and associated error mechanisms in symptom detection, offering insights into the challenge of improving reliability in global disease surveillance systems across diverse languages.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Overview of the study. We evaluate the capability of large language models to predict multilabel symptoms from social media posts across 12 languages, ranging from high-resource to low-resource languages and covering multiple regions. The left panel shows the same social media post across the languages, grouped by language resource level, with the gold symptom labels shown for each text. Bold-highlighted words represent symptom-related words. The right panel shows the model predictions for each symptom and language.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e103321_fig01.png"/></fig></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Dataset</title><p>This study uses MedWeb, a multilingual pseudo&#x2013;social media text dataset with multiple symptom labels. The dataset was originally created in Japanese and then human-translated into other languages. The translation-based dataset provides a controlled setting for consistent cross-linguistic comparison of symptom expression. It covers 12 languages, predominantly languages from Southeast Asia (Indonesian, Filipino, Malay, Thai, Khmer, Burmese, and Lao), together with Japanese, English, German, French, and Arabic. Each language contains 640 texts, which are labeled as positive or negative for each symptom or disease (hereafter referred to simply as symptom), allowing multiple positive labels per text [<xref ref-type="bibr" rid="ref20">20</xref>].</p><p>In this study, the languages were categorized by region and resource classification, as shown in <xref ref-type="table" rid="table1">Table 1</xref> [<xref ref-type="bibr" rid="ref6">6</xref>]. The language resource classification was included because LLM performance in NLP tasks is partly influenced by the availability of training data. Therefore, we also investigated whether performance on this symptom detection task differs across languages with varying resource levels.</p><p>The study focuses on 8 common symptoms, which are considered indicators associated with other diseases: fever, headache, runny nose, cough, diarrhea, hay fever, influenza, and cold. The label distribution in this dataset is imbalanced between positive and negative labels, with the proportion of positive labels varying across symptoms, as presented in <xref ref-type="table" rid="table2">Table 2</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Language resource classification.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Resource classification</td><td align="left" valign="bottom">Languages</td></tr></thead><tbody><tr><td align="left" valign="top">High</td><td align="left" valign="top">English, German, French, Japanese, and Arabic</td></tr><tr><td align="left" valign="top">Mid</td><td align="left" valign="top">Indonesian, Filipino, Malay, and Thai</td></tr><tr><td align="left" valign="top">Low</td><td align="left" valign="top">Lao, Khmer, and Burmese</td></tr></tbody></table></table-wrap><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Positive label distribution.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Symptoms</td><td align="left" valign="bottom">Number of texts with positive labels (N=640)<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">Influenza</td><td align="left" valign="top">24 (3.75)</td></tr><tr><td align="left" valign="top">Diarrhea</td><td align="left" valign="top">64 (10)</td></tr><tr><td align="left" valign="top">Hay fever</td><td align="left" valign="top">46 (7.19)</td></tr><tr><td align="left" valign="top">Cough</td><td align="left" valign="top">80 (12.5)</td></tr><tr><td align="left" valign="top">Headache</td><td align="left" valign="top">77 (12.03)</td></tr><tr><td align="left" valign="top">Fever</td><td align="left" valign="top">93 (14.53)</td></tr><tr><td align="left" valign="top">Runny nose</td><td align="left" valign="top">123 (19.22)</td></tr><tr><td align="left" valign="top">Cold</td><td align="left" valign="top">90 (14.06)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Percentages are calculated using the total number of texts (N=640) as the denominator for each symptom. As the dataset is multilabel, a text may have zero, one, or multiple positive symptom labels; therefore, label counts and percentages across symptoms are not expected to sum to N or 100%.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-2"><title>Experimental Settings</title><p>We used the GPT-5 (OpenAI) model as a tool for this study. A previous work showed that models from the OpenAI family consistently outperformed others across both large-parameter and small-parameter settings [<xref ref-type="bibr" rid="ref13">13</xref>]. In addition, GPT-5 is the strongest OpenAI model for health-related questions, achieving higher scores on HealthBench than previous models evaluated in that benchmark [<xref ref-type="bibr" rid="ref21">21</xref>]. For the prompting strategy, we implemented a rule-based, zero-shot setting in English to reflect a general and realistic deployment scenario. The rules incorporated into the prompt were aligned with the annotation guidelines used during dataset labeling, as shown in <xref ref-type="other" rid="box1">Textbox 1</xref>.</p><boxed-text id="box1"><title> The prompt used in this study.</title><p><bold>Instruction:</bold></p><p>Determine if the sender of this Twitter message is exhibiting symptoms for each of the following: influenza, diarrhea, hay fever, cough, headache, fever, runny nose, and cold. For each symptom, only answer either 0 or 1 for negative (no symptoms) or positive (has symptoms) respectively. Determination of symptoms is carried out based on the following rules:</p><list list-type="bullet"><list-item><p>Cases where the symptom is expressed directly, including mild symptoms, are considered positive.</p></list-item><list-item><p>A symptom can be labeled positive with indirect expressions of a symptom.</p></list-item><list-item><p>If a symptom is mentioned but then also dismissed or denied, this information is regarded as positive.</p></list-item><list-item><p>It is considered positive if someone or the user is still affected with such mild symptoms during recovery. However, if the symptoms are completely gone, it is considered negative.</p></list-item><list-item><p>A symptom is positive even if the user expresses uncertainty regarding its cause.</p></list-item><list-item><p>Since it is generally presumed that many patients may overlook symptoms or diseases due to insufficient medical knowledge, even suspicion of symptoms and diseases are recognized and labeled positive.</p></list-item><list-item><p>Symptoms that disappeared completely are recognized and labeled negative. Note that we regarded and labeled positive when a user took medicine that could cause temporary recovery from a symptom.</p></list-item><list-item><p>For cases that express expectation or process, indicated with words such as &#x201C;if,&#x201D; &#x201C;going,&#x201D; &#x201C;if it is,&#x201D; etc, these should be labeled as negative.</p></list-item><list-item><p>If the disease is mentioned merely as a topic rather than someone having it, these tweets should be labeled as negative. These include news, general theories, and advertisements.</p></list-item><list-item><p>If the disease is mentioned in the context of a joke, these should be labeled as negative.</p></list-item><list-item><p>The symptoms are only for humans.</p></list-item><list-item><p>Symptoms are within 24 hours, including today.</p></list-item><list-item><p>The label for symptoms that occurred yesterday is dependent on the disease or symptom.</p></list-item><list-item><p>Past symptoms, including symptoms 2 or more days ago, are considered negative.</p></list-item><list-item><p>Recent occurrences and recurring symptoms that still persist are considered positive.</p></list-item><list-item><p>We regard symptoms in the vicinity and label them as positive regardless of living together or not (ie, family members). We also label symptoms as positive when they are observed from hearsay.</p></list-item><list-item><p>As for symptoms of people belonging to a specified group in the vicinity (school, club, etc), we labeled them positive.</p></list-item><list-item><p>Other cases, like symptoms belonging to blog friends, should be labeled as negative since it is difficult to determine their location.</p></list-item></list><p><bold>Post:</bold></p><p>A post in one of the twelve studied languages.</p><p><bold>Output:</bold></p><p>Return the result strictly as a JSON object with the symptoms as keys and the values as either 0 or 1.</p></boxed-text><p>We also conducted an ablation study to support the selection of the LLM and the prompting strategy used in the main experiments. For model selection, we evaluated 1 proprietary LLM (GPT-5) and 3 groups of open-weight LLMs: (1) general-purpose models, including Gemma 3 4B (Google DeepMind), Qwen3-VL 8B (Qwen team, Alibaba Cloud), and Llama 3.1 8B (Meta); (2) models fine-tuned for Southeast Asian (SEA) languages, including Gemma-SEA-LION-v4-4B-VL (AI Singapore), Qwen-SEA-LION-v4-8B-VL (AI Singapore), and Llama-SEA-LION-v3-8B (AI Singapore); and (3) models fine-tuned for the medical context, including MedGemma 4B, Bio-Medical-Llama-3-8B, and HuatuoGPT-o1-7B. These models were selected to examine multilingual performance patterns across different model architectures and families and to identify the model with the most suitable performance for further analysis in this study.</p><p>In addition to open-weight LLMs, we included supervised models trained on the study dataset and a multilingual zero-shot architecture that was not fine-tuned on the dataset as baselines. These models represented both general-purpose multilingual and medical-domain multilingual approaches across different model paradigms. We further conducted a prompting-strategy ablation to examine the effect of alternative prompt designs. The evaluated strategies included five prompts: (1) rule-based zero-shot setting in English, (2) rule-based zero-shot prompting in each language, (3) rule-based few-shot prompting in English, (4) basic zero-shot prompting in English, and (5) basic zero-shot prompting in each language. This analysis enabled us to assess the contribution of prompt language and structure to model performance.</p><p>We further conducted a prompting-strategy ablation to examine the effect of alternative prompt designs. The evaluated strategies included five prompts: (1) rule-based zero-shot setting in English, (2) rule-based zero-shot prompting in each language, (3) rule-based few-shot prompting in English, (4) basic zero-shot prompting in English, and (5) basic zero-shot prompting in each language. This analysis enabled us to assess the contribution of prompt language and structure to model performance.</p><p>The complete experimental configurations, including the corresponding results for the evaluated models and prompting strategies, are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-3"><title>Evaluation Methods</title><p>In this study, we analyzed the results from 2 perspectives: a language-based and a symptom-based analysis. Both perspectives used macro precision, recall, and <italic>F</italic><sub>1</sub>-score. In the symptom-based analysis, the macro <italic>F</italic><sub>1</sub>-score assigns equal importance to each symptom label, regardless of whether the label is frequent or rare, by calculating the <italic>F</italic><sub>1</sub>-score for each label separately and then averaging them [<xref ref-type="bibr" rid="ref22">22</xref>]. This metric provides a clearer picture of the model&#x2019;s consistency and enables analysis of performance variation across symptoms. Meanwhile, recall measures how many true positive symptom labels the model successfully captures on average for each symptom. In the context of surveillance, recall is important because low recall indicates that the system is missing many true symptom signals. Precision, on the other hand, measures the proportion of predicted positive symptom labels that are correct. In a surveillance system, this is useful for identifying which symptoms generate more false positives, potentially leading to false alarms. This study also conducted an error analysis to identify the mechanisms underlying incorrect symptom predictions across languages. An error tweet was defined as a tweet for which (1) at least one language version contained an incorrect prediction and (2) the prediction patterns were not identical across the language versions. Based on the identified error tweets, we first performed a manual qualitative review to identify recurring error patterns and grouped them into several categories. We then defined each error type and prepared representative examples. Using these definitions and examples, an LLM (GPT-5-mini) was asked to classify all identified error tweets into the corresponding error categories and assign a primary error category to each tweet. Approximately 5% of the LLM-generated classifications were subsequently manually reviewed by a human for quality control. The counterfactual impact analysis was then done by correcting the incorrect predictions in each error tweet based on its primary error category and recalculating the <italic>F</italic><sub>1</sub>-score. The macro <italic>F</italic><sub>1</sub>-score was recalculated on the complete multilingual dataset (7680 language-specific texts), with <italic>F</italic><sub>1</sub>-score calculated for each symptom across all languages and then macroaveraged across the 8 symptom labels. The residual impact of each error type (&#x0394;F<sub>1</sub>) was then measured as the difference between the corrected and original <italic>F</italic><sub>1</sub>-scores (baseline). A smaller &#x0394;F<sub>1</sub> means that the error category is causing less remaining performance loss under that prompt.</p></sec><sec id="s2-4"><title>Ethical Considerations</title><p>This study involved no physical or mental interventions and did not include any experiments requiring human participation. In accordance with the Ethical Guidelines for Medical and Biological Research Involving Human Subjects established by the Japanese government, this study did not require institutional review board approval, as no personally identifiable information was used [<xref ref-type="bibr" rid="ref23">23</xref>]. The dataset used in this study is publicly available as the extended MedWeb corpus. The original study about the dataset also states that it does not contain personally identifiable information and is exempt from institutional review board approval [<xref ref-type="bibr" rid="ref20">20</xref>]. Therefore, this study raises no ethical concerns with respect to user privacy or informed consent.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Study Findings</title><p>Before presenting the main findings, we conducted ablation studies to select the model and prompting strategy. As shown in Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, among the evaluated LLMs, GPT-5 achieved the highest overall performance with relatively low disparities across languages. In Figure S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, among prompting strategies, rule-based zero-shot prompting in English showed the lowest performance variability across languages, although its average performance was lower than that of some alternative prompting strategies. Based on these findings, we selected GPT-5 with rule-based zero-shot prompting in English for the main analysis to ensure high and consistent performance across languages.</p><p>This section presents the evaluation results of LLM-based multilabel symptom detection across 12 languages, with the aim of examining performance disparities among the languages. The results were analyzed from 2 perspectives: language-level performance and symptom-level performance.</p><p>The results of the language-based analysis, as presented in <xref ref-type="table" rid="table3">Table 3</xref>, showed a noticeable performance gap between SEA languages and the other languages. Overall, SEA languages tend to achieve lower scores across macro <italic>F</italic><sub>1</sub>-score, precision, and recall than non-SEA languages, with Thai being the only exception. Thai achieved the strongest performance among SEA languages, with scores approaching those of other high-resource languages such as Japanese, English, and French. Among SEA languages, Lao had the lowest <italic>F</italic><sub>1</sub>-score (0.716) and the widest gap between precision (0.820) and recall (0.658). This indicates that although the model&#x2019;s predictions are relatively precise, it fails to identify many actual positive cases in the Lao dataset.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Overall performance metrics across languages, sorted by <italic>F</italic><sub>1</sub>-score from the highest to lowest<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Language</td><td align="left" valign="bottom">Resource classification</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td></tr></thead><tbody><tr><td align="left" valign="top">Japanese</td><td align="left" valign="top">High</td><td align="char" char="." valign="top">0.812</td><td align="char" char="." valign="top">0.888</td><td align="char" char="." valign="top">0.770</td></tr><tr><td align="left" valign="top">English</td><td align="left" valign="top">High</td><td align="char" char="." valign="top">0.791</td><td align="char" char="." valign="top">0.873</td><td align="char" char="." valign="top">0.753</td></tr><tr><td align="left" valign="top">French</td><td align="left" valign="top">High</td><td align="char" char="." valign="top">0.788</td><td align="char" char="." valign="top">0.869</td><td align="char" char="." valign="top">0.750</td></tr><tr><td align="left" valign="top">Arabic</td><td align="left" valign="top">High</td><td align="char" char="." valign="top">0.784</td><td align="char" char="." valign="top">0.878</td><td align="char" char="." valign="top">0.736</td></tr><tr><td align="left" valign="top">German</td><td align="left" valign="top">High</td><td align="char" char="." valign="top">0.783</td><td align="char" char="." valign="top">0.873</td><td align="char" char="." valign="top">0.737</td></tr><tr><td align="left" valign="top">Thai</td><td align="left" valign="top">Mid</td><td align="char" char="." valign="top">0.780</td><td align="char" char="." valign="top">0.857</td><td align="char" char="." valign="top">0.742</td></tr><tr><td align="left" valign="top">Malay</td><td align="left" valign="top">Mid</td><td align="char" char="." valign="top">0.735</td><td align="char" char="." valign="top">0.814</td><td align="char" char="." valign="top">0.699</td></tr><tr><td align="left" valign="top">Filipino</td><td align="left" valign="top">Mid</td><td align="char" char="." valign="top">0.731</td><td align="char" char="." valign="top">0.802</td><td align="char" char="." valign="top">0.684</td></tr><tr><td align="left" valign="top">Indonesian</td><td align="left" valign="top">Mid</td><td align="char" char="." valign="top">0.731</td><td align="char" char="." valign="top">0.807</td><td align="char" char="." valign="top">0.689</td></tr><tr><td align="left" valign="top">Khmer</td><td align="left" valign="top">Low</td><td align="char" char="." valign="top">0.728</td><td align="char" char="." valign="top">0.828</td><td align="char" char="." valign="top">0.692</td></tr><tr><td align="left" valign="top">Burmese</td><td align="left" valign="top">Low</td><td align="char" char="." valign="top">0.726</td><td align="char" char="." valign="top">0.826</td><td align="char" char="." valign="top">0.682</td></tr><tr><td align="left" valign="top">Lao</td><td align="left" valign="top">Low</td><td align="char" char="." valign="top">0.716</td><td align="char" char="." valign="top">0.820</td><td align="char" char="." valign="top">0.658</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>All metrics indicate that Southeast Asian (SEA) languages tend to perform lower than the others.</p></fn></table-wrap-foot></table-wrap><p>This finding is consistent with the ablation study in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> generally shows lower performance for SEA languages than for non-SEA languages across LLMs, except for the supervised models. Moreover, this performance gap persists even in models specifically fine-tuned for SEA languages.</p><p>To further investigate the disparities, we analyze the performance differences across language resource classification groups, as shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>. High-resource languages achieved the highest and the most consistent performance, with Japanese achieving the highest score and appearing as an outlier. The midresource group exhibited greater variability, with Thai standing out as a high-performing outlier, as its performance is closest to that of the high-resource group, while the remaining languages within the group show relatively similar performance levels. Low-resource languages showed the lowest performance among all groups, with Lao obtaining the lowest <italic>F</italic><sub>1</sub>-score among languages. These results indicate performance disparities across the resource language classification.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Distribution of average <italic>F</italic><sub>1</sub>-score across language resource categories (High, mid, and low). Labeled data points represent observations that are potential outliers within each category. High-resource languages achieved the highest performance, while midresource and low-resource languages showed lower performance, with low-resource languages obtaining the lowest overall results.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e103321_fig02.png"/></fig><p>These results align with findings from our model ablation study in Figure S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. High-resource languages generally achieve better performance than low-resource languages across the evaluated LLMs, although the ranking of individual languages varies by model category. Japanese achieves the highest performance on GPT-5, while English consistently performs best among high-resource languages across the 3 open-weight LLM groups and the zero-shot pretrained classifier.</p><p>Furthermore, we analyzed symptom-level data across languages. <xref ref-type="fig" rid="figure3">Figure 3</xref> shows that detection performance varies substantially across symptoms, with average macro <italic>F</italic><sub>1</sub>-scores ranging from 0.560 (SD 0.046) for runny nose to 0.896 (SD 0.017) for diarrhea. This suggests that the model&#x2019;s ability to detect symptoms differs across symptom types. Beyond overall symptom-level differences, we further examine how performances vary across languages within each symptom.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Average macro <italic>F</italic><sub>1</sub>-score for each symptom across languages, grouped by resource classification and region. High-resource languages tend to achieve higher scores across most symptoms, while greater performance variation is observed in midresource and low-resource languages, particularly for hay fever, where Southeast Asian (SEA) languages show the widest range of scores.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ai_v5i1e103321_fig03.png"/></fig><p>Specifically, there is a notable variation in performance across languages for several symptoms. Diarrhea, headache, cough, and fever showed relatively stable performance across languages. However, cold, runny nose, influenza, and hay fever exhibit large performance variations. Most notably, hay fever detection reveals 2 distinct performance clusters. The first cluster shows high scores, with an average macro <italic>F</italic><sub>1</sub>-score of 0.865 (SD 0.024). The second cluster exhibits lower and more varied performance, with an average macro <italic>F</italic><sub>1</sub>-score of 0.607 (SD 0.105). All languages in the low-performance cluster are SEA languages, and these include midresource and low-resource languages. In contrast, the high-performance cluster consists predominantly of high-resource languages, with Thai being the only SEA language in this group. Despite this variation, a consistent pattern emerges that SEA languages classified as midresource to low-resource tend to be placed lower. This indicates a performance gap between languages for some symptoms.</p><p>However, an interesting pattern emerged from the experiments with different prompting strategies, as shown in Figure S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Few-shot prompting appears to improve the model&#x2019;s understanding of certain symptoms, particularly hay fever, fever, and runny nose, while no significant differences are observed for the other symptoms. Compared with the basic prompt, however, some symptoms show decreased performance, including diarrhea, hay fever, cough, fever, influenza, and runny nose.</p><p>Furthermore, we analyzed precision and recall at the symptom level to understand how the balance between these 2 metrics shapes implications for global disease surveillance systems, particularly for potential outbreak detection. <xref ref-type="table" rid="table4">Table 4</xref> showed that several symptoms exhibit an imbalance between their recall and precision scores, with most symptoms having lower recall compared to their precision. Runny nose demonstrated the greatest disparity between the 2, with recall considerably low at 0.427 and precision relatively high at 0.858. Meanwhile, influenza was the only symptom where precision scored lower than recall, indicating that the model tends to overpredict influenza.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Recall and precision score at the symptom level.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Symptoms</td><td align="left" valign="bottom">Average of recall (SD)</td><td align="left" valign="bottom">Average of precision (SD)</td></tr></thead><tbody><tr><td align="left" valign="top">Diarrhea</td><td align="char" char="." valign="top">0.857 (0.023)</td><td align="char" char="." valign="top">0.940 (0.017)</td></tr><tr><td align="left" valign="top">Headache</td><td align="char" char="." valign="top">0.903 (0.019)</td><td align="char" char="." valign="top">0.851 (0.014)</td></tr><tr><td align="left" valign="top">Cold</td><td align="char" char="." valign="top">0.742 (0.054)</td><td align="char" char="." valign="top">0.906 (0.066)</td></tr><tr><td align="left" valign="top">Cough</td><td align="char" char="." valign="top">0.645 (0.023)</td><td align="char" char="." valign="top">0.968 (0.017)</td></tr><tr><td align="left" valign="top">Hay fever</td><td align="char" char="." valign="top">0.685 (0.209)</td><td align="char" char="." valign="top">0.826 (0.057)</td></tr><tr><td align="left" valign="top">Fever</td><td align="char" char="." valign="top">0.667 (0.052)</td><td align="char" char="." valign="top">0.797 (0.037)</td></tr><tr><td align="left" valign="top">Influenza</td><td align="char" char="." valign="top">0.806 (0.129)</td><td align="char" char="." valign="top">0.613 (0.059)</td></tr><tr><td align="left" valign="top">Runny nose</td><td align="char" char="." valign="top">0.427 (0.059)</td><td align="char" char="." valign="top">0.858 (0.149)</td></tr></tbody></table></table-wrap></sec><sec id="s3-2"><title>Error Analysis</title><sec id="s3-2-1"><title>Overview of Error Patterns and Impact</title><p>In analyzing symptom performance variability, we observed several recurring error patterns. This qualitative error analysis began by examining prediction patterns in a subsample of the dataset. The identified errors were then grouped into four categories: (1) explicitly mentioned symptoms, (2) cross-lingual variation, (3) symptom overgeneralization, and (4) context misinterpretation. Each error type is illustrated using representative tweets and their corresponding prediction errors, as shown in Table S3 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. An LLM was subsequently used to classify the identified error tweets into 1 or more of these categories, and a counterfactual impact analysis was subsequently conducted for each primary error category.</p><p>Counterfactual impact analysis was conducted across different prompting strategies, all using rule-based prompts, to examine how prompt design affects error patterns and their impact on performance. As shown in <xref ref-type="table" rid="table5">Table 5</xref>, few-shot prompting in English achieved the greatest improvement. Compared with zero-shot prompting in English, the baseline macro <italic>F</italic><sub>1</sub>-score increased by 0.0486, from 0.7613 to 0.8099, while the number of error tweets decreased from 296 to 270, a reduction of 26 (8.78%) cases. In contrast, zero-shot prompting in the local language produced only a modest improvement in overall performance.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Counterfactual impact of primary error categories across rule-based prompting strategies<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Prompt</td><td align="left" valign="bottom">Error tweets</td><td align="left" valign="bottom">Baseline <italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom" colspan="4">Corrected <italic>F</italic><sub>1</sub>-score (&#x0394;F<sub>1</sub>)</td></tr></thead><tbody><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">Explicitly mentioned symptoms</td><td align="left" valign="top">Cross-lingual variation</td><td align="left" valign="top">Symptom overgeneralization</td><td align="left" valign="top">Context misinterpretation</td></tr><tr><td align="left" valign="top">Rule-zero-English</td><td align="left" valign="top">296</td><td align="left" valign="top">0.7613</td><td align="left" valign="top">0.8547 (+0.0934)</td><td align="left" valign="top">0.8085 (+0.0472)</td><td align="left" valign="top">0.7738 (+0.0125)</td><td align="left" valign="top">0.8023 (+0.0410)</td></tr><tr><td align="left" valign="top">Rule-based few-shot English</td><td align="left" valign="top">270</td><td align="left" valign="top">0.8099</td><td align="left" valign="top">0.8531 (+0.0432)</td><td align="left" valign="top">0.8527 (+0.0428)</td><td align="left" valign="top">0.8206 (+0.0107)</td><td align="left" valign="top">0.8454 (+0.0355)</td></tr><tr><td align="left" valign="top">Rule-zero-local</td><td align="left" valign="top">293</td><td align="left" valign="top">0.7707</td><td align="left" valign="top">0.8515 (+0.0808)</td><td align="left" valign="top">0.8095 (+0.0388)</td><td align="left" valign="top">0.7782 (+0.0075)</td><td align="left" valign="top">0.8035 (+0.0328)</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>Values in parentheses indicate the change in macro <italic>F</italic><sub>1</sub>-score (&#x0394;<italic>F</italic><sub>1</sub>) after correcting prediction errors in tweets assigned to each primary error category.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2-2"><title>Explicitly Mentioned Symptoms</title><p>Errors in this category occur when a model&#x2019;s predictions depend on whether a symptom is explicitly expressed in the text. For example, in Case ID 2512, a tweet labeled as influenza and fever is predicted only as influenza in the English dataset, indicating a missed symptom. In contrast, in the Malay version, fever is detected due to the presence of the word <italic>demam</italic>, which corresponds to fever in English. This example shows that the model&#x2019;s predictions are related to the terms explicitly mentioned in the text.</p><p>Based on the error impact analysis, explicitly mentioned symptoms had the greatest impact on model performance across all 3 prompting strategies. Under zero-shot prompting in English, correcting this error category resulted in a +0.0934 increase in macro <italic>F</italic><sub>1</sub>-score, which was nearly twice the impact of cross-lingual variation or context misinterpretation. With few-shot prompting, the potential impact decreased to +0.0432<italic>;</italic> meanwhile<italic>,</italic> under zero-shot prompting in the local language, it remained relatively high at +0.0808. These findings suggest that providing examples in the prompt may help the model recognize symptom expressions more effectively than simply using the local language for the detection instructions.</p></sec><sec id="s3-2-3"><title>Cross-Lingual Variation</title><p>This error type arises from cross-lingual variation, as differences in how symptoms are expressed across languages can lead to different prediction outcomes. <italic>Demam selesema</italic> in Malay can refer to influenza or a cold, which may lead to confusion between these labels. Another example is shown in case ID 2045, where a tweet labeled as cold is predicted as runny nose in the Filipino and Indonesian datasets. This occurs because <italic>sipon</italic> in Filipino and <italic>pilek</italic> in Indonesian can refer to both runny nose and cold, contributing to misclassification.</p><p>Cross-lingual variation was the most frequent error type across all rule-based prompting strategies. However, the impact analysis showed that changing the prompting strategy only modestly reduced its impact. The potential impact of cross-lingual variation was slightly lower with local-language prompting (+0.0388) than with English zero-shot prompting (+0.0472), suggesting that local-language instructions may partially reduce language-specific errors but do not fully address cross-lingual variation.</p></sec><sec id="s3-2-4"><title>Symptom Overgeneralization</title><p>In addition to errors arising from the terms used in the text, the model also shows limitations in understanding context and in disease-related or symptom-related concepts. This is reflected in cases of symptom overgeneralization, where the model predicts additional symptoms commonly associated with the expressed condition but not explicitly stated. For example, in case ID 2119, a tweet labeled as cold and headache is additionally predicted as fever in both the English and Malay datasets.</p><p>Based on the impact analysis, this error category had a smaller effect on overall <italic>F</italic><sub>1</sub>-score performance than the other 3 categories. This is reflected in the consistently smallest difference between the corrected and baseline <italic>F</italic><sub>1</sub>-scores across all evaluated prompting strategies. Moreover, the relatively similar impact across prompting conditions suggests that modifying the prompt alone may have limited influence on this type of error.</p></sec><sec id="s3-2-5"><title>Context Misinterpretation</title><p>Moreover, context misinterpretation is observed when the model predicts a symptom based solely on its mention without considering the surrounding context. For example, in case ID 2165, the word &#x201C;flu&#x201D; appears in the text, but the overall context indicates that the individual does not have influenza. In the English and Arabic datasets, the model correctly predicts the absence of influenza, but in the German and French datasets, it incorrectly predicts the presence of influenza.</p><p>The impact analysis showed that contextual errors persisted across all prompting strategies. This suggests that neither providing examples in the prompt nor using the local language was sufficient to fully address context misinterpretation.</p></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><sec id="s4-1-1"><title>Performance Overview</title><p>This study examined the performance gaps in symptom detection tasks using a general-purpose LLM. Based on the 2 perspectives of analysis, language-based and symptom level, our findings showed that performance varies across languages, their resource classifications, and symptom levels.</p></sec><sec id="s4-1-2"><title>Language-Based Analysis</title><p>In this study, language-based disparities were observed between SEA and non-SEA languages. This regional pattern reflects the languages included in each group and their corresponding resource availability rather than indicating that the geographic region itself determines model performance. Most SEA languages in this study are classified as midresource to low-resource, while most non-SEA languages are high-resource. Previous studies have reported similar results regarding performance disparities between English and non-English languages in medical applications of LLMs [<xref ref-type="bibr" rid="ref24">24</xref>-<xref ref-type="bibr" rid="ref26">26</xref>], indicating that multilingual performance differences are closely related to how well individual languages are represented and supported in the model.</p><p>This interpretation is also consistent with the performance of individual languages. In the GPT-5 experiment, Japanese achieved the highest performance, which may reflect the fact that the Japanese dataset is the original version and therefore preserves the source linguistic context without translation. In contrast, English consistently achieved the highest performance across the other open-weight LLMs. This suggests that the relative advantage of Japanese or English is model-dependent, where Japanese may benefit from being the source language of the dataset and English may benefit from stronger representation in the multilingual training data of some models. Nevertheless, both languages generally outperform the midresource and low-resource languages.</p><p>More broadly, previous studies have shown that multilingual LLM performance can be influenced by factors such as pretraining data size and the availability of language resources [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. This is consistent with our findings, which show that high-resource languages generally outperform low-resource languages. Since region and resource level overlap in this study, the lower performance observed for SEA languages is better interpreted as a language-related and resource-related disparity rather than a regional effect. This distinction is essential because SEA is characterized by substantial linguistic diversity and includes many languages with relatively limited NLP resources, and the region is also highly relevant to infectious disease surveillance.</p></sec><sec id="s4-1-3"><title>Symptom-Based Analysis</title><p>We further examine performance differences across symptoms. Our findings indicate that performance varies by symptom, with some symptoms showing relatively consistent performance across languages, while others exhibit substantial variation. Symptoms with low cross-lingual variation, such as diarrhea, headache, cough, and fever, indicate that they are consistently expressed and identified easily across languages. Their lower cross-lingual variability may reflect more consistent symptom expressions or easier model recognition across the evaluated languages.</p><p>In contrast, symptoms such as cold, hay fever, influenza, and runny nose exhibit greater performance variability, indicating less stable detection across languages. The model may identify these symptoms well in some languages but not in others. Additionally, compared to low-variation symptoms, these symptoms often overlap with other related symptoms, making them more difficult to detect [<xref ref-type="bibr" rid="ref29">29</xref>-<xref ref-type="bibr" rid="ref31">31</xref>]. From a linguistic perspective, this variability also suggests that model performance may depend on the availability of training data and the familiarity with the symptom or disease concept within each language.</p><p>Moreover, the error analysis identifies the types of symptom misclassifications. The results indicate that the model relies on lexical cues in the text, while cross-lingual variation in terminology is associated with differences in prediction outcomes. Beyond these errors, the model also predicts additional symptoms associated with the expressed condition even when they are not explicitly mentioned and misclassifies cases in which symptom-related terms are present but negated by the broader context. These findings indicate that the model has difficulty handling implicit meaning, language-specific symptom expressions, and understanding disease concepts.</p><p>In terms of frequency, cross-lingual variation was the most common error type. However, the impact analysis showed that the greatest potential performance loss was due to the model&#x2019;s difficulty in interpreting implicit symptom expressions. Few-shot prompting appeared to reduce both types of errors, particularly the explicitly mentioned symptom errors, while zero-shot prompting in the local language produced only limited improvement. These findings suggest that prompt design alone may not be sufficient to fully address the challenges of multilingual symptom detection.</p></sec></sec><sec id="s4-2"><title>Interpreting Hay Fever Variability</title><p>Since the primary focus of this study was to analyze disparities in symptom detection performance, we further examined hay fever as a representative symptom, as its detection performance appeared to vary substantially across languages. Hay fever formed 2 distinct clusters of high-performing and low-performing languages, with most SEA languages grouped in the lower-performing cluster. This pattern suggests that some languages support more accurate detection of hay fever, while others may be less capable of capturing hay fever&#x2013;related terminology or symptom expressions.</p><p>This finding emerges alongside regional differences in hay fever&#x2013;related research and reported cases. Hay fever or allergic rhinitis has been more frequently reported and discussed in regions such as Europe, North America, the Mediterranean, and Japan, while related studies and reported cases remain limited in Southeast Asia [<xref ref-type="bibr" rid="ref32">32</xref>-<xref ref-type="bibr" rid="ref36">36</xref>]. As a result, hay fever may be more clearly defined and more frequently referenced in the languages of those regions compared with those used in Southeast Asia. Meanwhile, in LLM-based systems, LLM training data likely reflect the availability of published literature and web content. Therefore, symptoms that are underrepresented in certain language text corpora will be less recognized by the model, causing those symptoms to be more difficult to predict consistently.</p><p>This interpretation is also consistent with the ablation study, where providing the model with examples of hay fever cases through few-shot prompting substantially improved its detection performance. This suggests that increasing the model&#x2019;s familiarity with symptom-related expressions can improve its ability to recognize hay fever across languages. Therefore, the lower performance observed in some languages could be associated with limited exposure to hay fever&#x2013;related terminology and expressions.</p></sec><sec id="s4-3"><title>Implications for Global Disease Surveillance</title><p>This study has important implications for the use of LLMs in public health surveillance. Symptom detection is a crucial step in digital surveillance systems, as it enables the identification of health-related signals. However, in global applications involving multiple languages, language-related performance disparities may prevent the model from capturing signals consistently across populations. For example, our findings showed that the model achieved low recall for Lao, indicating that many actual symptom labels in the Lao dataset were missed. As a result, the symptoms from that population could be underdetected. This could lead to certain populations being underrepresented in monitoring and analysis, particularly among populations that are vulnerable to emerging infectious diseases.</p><p>In addition to language disparities, performance variation was also observed across symptom types. Large performance differences across symptoms may lead to false alarms or missed signals in surveillance systems. For example, based on our findings, runny nose showed the largest gap between precision and recall, with relatively low recall, suggesting that the system may fail to capture many true runny nose signals. In contrast, influenza was the only symptom for which precision was lower than recall, indicating a greater tendency toward false positive predictions. This may result in influenza-related signals being overdetected in regions without actual outbreaks, potentially leading to unnecessary public concern and inefficient allocation of public health resources.</p><p>Furthermore, our findings suggest that the model tends to rely on lexical cues in symptom-related text, while still requiring contextual understanding of diseases and the ability to handle diverse symptom expressions. Therefore, LLM-based symptom-detection systems used in public health surveillance should account for these limitations to ensure consistent and reliable performance. This is particularly important when deploying such systems across diverse linguistic and epidemiological contexts.</p></sec><sec id="s4-4"><title>Limitations</title><p>This study has several limitations. First, although many symptoms commonly experienced by individuals may serve as signals of other diseases, this study included only 8 symptoms and diseases as representatives. Second, this study uses a translation-based multilingual dataset, which may not fully reflect natural language use in each target language, even though the translations were performed by native speakers. Third, this study uses a limited set of languages and includes Southeast Asia as a representative region that is vulnerable to emerging infectious diseases. However, we acknowledge that other regions may also serve as potential origin points for future outbreaks. Expanding the analysis to a broader range of geographical regions would provide a more comprehensive understanding of global disparities. Fourth, the error analysis is based on qualitative methods in the initial observations, which may not capture all possible sources of model failure. Further research should address these limitations to provide a more comprehensive understanding of potential disparities in symptom detection performance for public health surveillance systems.</p></sec><sec id="s4-5"><title>Conclusions</title><p>This study demonstrates the presence of multilingual performance disparities in LLM-based symptom detection and offers insights into developing a reliable global disease surveillance system that operates across diverse languages. We identified 3 main points from the study&#x2019;s findings. First, detection performance tends to vary with language resource level. The observed performance gap between high-resource and low-resource languages, which could be associated with differences in language resource availability, could contribute to underdetection of symptoms in linguistically underrepresented populations, particularly those who are vulnerable to emerging infectious diseases. Second, not all symptoms are equally detectable across languages. While some symptoms are reliably detected across languages, others remain inconsistently identified due to linguistic variability, symptom overlap, and training data familiarity with the symptoms. Third, the error patterns suggest limitations in the model&#x2019;s ability to detect symptoms from social media text. The current model&#x2019;s tendency to rely on lexical cues, along with limited contextual and disease-level understanding, represents a challenge for the LLM-based symptom detection systems to identify consistently across diverse languages and symptom expressions. Few-shot prompting may improve the recognition of implicit symptom expressions; however, prompting alone does not fully address broader symptom understanding or contextual interpretation. Therefore, to achieve more reliable and equitable LLM-based symptom detection from social media text for global disease surveillance, it would benefit from broader representation of training data for low-resource languages, improved cultural-linguistic sensitivity, and stronger contextual understanding of symptom-related expressions.</p></sec></sec></body><back><ack><p>The authors declare the use of generative AI (GenAI) in the research and writing process. According to the GenAI delegation taxonomy [<xref ref-type="bibr" rid="ref37">37</xref>], the following tasks were delegated to GenAI tools under full human supervision: text generation, proofreading and editing, summarizing text, adapting and adjusting emotional tone, translation, code generation, and quality assessment. The GenAI tools used were ChatGPT-5.5, Claude Sonnet 4.6, and Grammarly. Text generation and summarizing text were used to generate the text, which was then revised and reviewed by the author to ensure that the core of the sentences matched what the human intended. Code generation was used to assist the human to create more effective code. Quality assessment was used to provide recommendations for improving the writing of the paper, while the remaining research thoughts and ideas were carried out by the authors. Responsibility for the final manuscript lies entirely with the authors. GenAI tools are not listed as authors and do not bear responsibility for the final outcomes.</p></ack><notes><sec><title>Funding</title><p>This work was supported by the Cross-ministerial Strategic Innovation Promotion Program (SIP) on the "Integrated Health Care System" (grant JPJ012425) and JST CREST (grant JPMJCR22N1).</p></sec><sec><title>Data Availability</title><p>The dataset used in this study is available in the MedWeb repository [<xref ref-type="bibr" rid="ref38">38</xref>].</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb2">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb3">SEA</term><def><p>Southeast Asian</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Krafft</surname><given-names>T</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Q</given-names> </name></person-group><article-title>Progress and challenges in infectious disease surveillance and early warning</article-title><source>Med Plus</source><year>2025</year><month>03</month><volume>2</volume><issue>1</issue><fpage>100071</fpage><pub-id pub-id-type="doi">10.1016/j.medp.2025.100071</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shi</surname><given-names>B</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Dang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>W</given-names> </name></person-group><article-title>Leveraging social media data for pandemic detection and prediction</article-title><source>Humanit Soc Sci Commun</source><year>2024</year><month>08</month><day>23</day><volume>11</volume><issue>1</issue><fpage>1075</fpage><pub-id pub-id-type="doi">10.1057/s41599-024-03589-y</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aiello</surname><given-names>AE</given-names> </name><name name-style="western"><surname>Renson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zivich</surname><given-names>PN</given-names> </name></person-group><article-title>Social media&#x2013;and internet-based disease surveillance for public health</article-title><source>Annu Rev Public Health</source><year>2020</year><month>04</month><day>2</day><volume>41</volume><fpage>101</fpage><lpage>118</lpage><pub-id pub-id-type="doi">10.1146/annurev-publhealth-040119-094402</pub-id><pub-id pub-id-type="medline">31905322</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Amin</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zeb</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Alshahrani</surname><given-names>H</given-names> </name><name name-style="western"><surname>Hamdi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Alsulami</surname><given-names>M</given-names> </name><name name-style="western"><surname>Shaikh</surname><given-names>A</given-names> </name></person-group><article-title>Social media-based surveillance systems for health informatics using machine and deep learning techniques: a comprehensive review and open challenges</article-title><source>Comput Model Eng Sci</source><year>2024</year><volume>139</volume><issue>2</issue><fpage>1167</fpage><lpage>1202</lpage><pub-id pub-id-type="doi">10.32604/cmes.2023.043921</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wilson</surname><given-names>AE</given-names> </name><name name-style="western"><surname>Lehmann</surname><given-names>CU</given-names> </name><name name-style="western"><surname>Saleh</surname><given-names>SN</given-names> </name><name name-style="western"><surname>Hanna</surname><given-names>J</given-names> </name><name name-style="western"><surname>Medford</surname><given-names>RJ</given-names> </name></person-group><article-title>Social media: a new tool for outbreak surveillance</article-title><source>Antimicrob Steward Healthc Epidemiol</source><year>2021</year><volume>1</volume><issue>1</issue><fpage>e50</fpage><pub-id pub-id-type="doi">10.1017/ash.2021.225</pub-id><pub-id pub-id-type="medline">36168466</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Joshi</surname><given-names>P</given-names> </name><name name-style="western"><surname>Santy</surname><given-names>S</given-names> </name><name name-style="western"><surname>Budhiraja</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bali</surname><given-names>K</given-names> </name><name name-style="western"><surname>Choudhury</surname><given-names>M</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Jurafsky</surname><given-names>D</given-names> </name><name name-style="western"><surname>Chai</surname><given-names>J</given-names> </name><name name-style="western"><surname>Schluter</surname><given-names>N</given-names> </name><name name-style="western"><surname>Tetreault</surname><given-names>J</given-names> </name></person-group><article-title>The state and fate of linguistic diversity and inclusion in the NLP world</article-title><source>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics</source><year>2020</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>6282</fpage><lpage>6293</lpage><pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.560</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Pava</surname><given-names>JN</given-names> </name><name name-style="western"><surname>Meinhardt</surname><given-names>C</given-names> </name><name name-style="western"><surname>Uz Zaman</surname><given-names>HB</given-names> </name><etal/></person-group><article-title>Mind the (language) gap: mapping the challenges of LLM development in low-resource language contexts</article-title><year>2025</year><access-date>2026-09-12</access-date><publisher-name>Stanford Institute for Human-Centered Artificial Intelligence (HAI), Stanford University</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://hai.stanford.edu/assets/files/hai-taf-pretoria-white-paper-mind-the-language-gap.pdf">https://hai.stanford.edu/assets/files/hai-taf-pretoria-white-paper-mind-the-language-gap.pdf</ext-link></comment></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Li</surname><given-names>S</given-names> </name><name name-style="western"><surname>Hauer</surname><given-names>B</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kondrak</surname><given-names>G</given-names> </name></person-group><article-title>Don&#x2019;t trust ChatGPT when your question is not in English: a study of multilingual abilities and types of LLMs</article-title><source>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing</source><year>2023</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>7915</fpage><lpage>7927</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.491</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Strasser</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Anschuetz</surname><given-names>W</given-names> </name><name name-style="western"><surname>Dennst&#x00E4;dt</surname><given-names>F</given-names> </name><name name-style="western"><surname>Hastings</surname><given-names>J</given-names> </name></person-group><article-title>Performance evaluation of large language models in multilingual medical multiple-choice questions: mixed methods study</article-title><source>JMIR Med Educ</source><year>2026</year><month>03</month><day>5</day><volume>12</volume><fpage>e81399</fpage><pub-id pub-id-type="doi">10.2196/81399</pub-id><pub-id pub-id-type="medline">41813244</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Deiner</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Deiner</surname><given-names>NA</given-names> </name><name name-style="western"><surname>Hristidis</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Use of large language models to assess the likelihood of epidemics from the content of tweets: infodemiology study</article-title><source>J Med Internet Res</source><year>2024</year><month>03</month><day>1</day><volume>26</volume><fpage>e49139</fpage><pub-id pub-id-type="doi">10.2196/49139</pub-id><pub-id pub-id-type="medline">38427404</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gallego</surname><given-names>F</given-names> </name><name name-style="western"><surname>Veredas</surname><given-names>FJ</given-names> </name></person-group><article-title>Recognition and normalization of multilingual symptom entities using in-domain-adapted BERT models and classification layers</article-title><source>Database (Oxford)</source><year>2024</year><month>08</month><day>28</day><volume>2024</volume><fpage>baae087</fpage><pub-id pub-id-type="doi">10.1093/database/baae087</pub-id><pub-id pub-id-type="medline">39197057</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yeh</surname><given-names>HS</given-names> </name><name name-style="western"><surname>Lavergne</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zweigenbaum</surname><given-names>P</given-names> </name></person-group><article-title>Challenges in multilingual adverse drug reaction detection on social media: insights from case studies</article-title><source>Stud Health Technol Inform</source><year>2025</year><month>08</month><day>7</day><volume>329</volume><fpage>450</fpage><lpage>454</lpage><pub-id pub-id-type="doi">10.3233/SHTI250880</pub-id><pub-id pub-id-type="medline">40775898</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Jannah</surname><given-names>SZ</given-names> </name><name name-style="western"><surname>Aco</surname><given-names>E</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wakamiya</surname><given-names>S</given-names> </name><name name-style="western"><surname>Aramaki</surname><given-names>E</given-names> </name></person-group><article-title>Multilingual symptom detection on social media: enhancing health-related fact-checking with LLMs</article-title><source>Proceedings of the Eighth Fact Extraction and VERification Workshop (FEVER)</source><year>2025</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>54</fpage><lpage>68</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.fever-1.4</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Susanto</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hulagadri</surname><given-names>AV</given-names> </name><name name-style="western"><surname>Montalan</surname><given-names>JR</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Che</surname><given-names>W</given-names> </name><name name-style="western"><surname>Nabende</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shutova</surname><given-names>E</given-names> </name><name name-style="western"><surname>Pilehvar</surname><given-names>MT</given-names> </name></person-group><article-title>SEA-HELM: Southeast Asian holistic evaluation of language models</article-title><source>Findings of the Association for Computational Linguistics</source><year>2025</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>12308</fpage><lpage>12336</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.findings-acl.636</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Ying</surname><given-names>J</given-names> </name><name name-style="western"><surname>Aljunied</surname><given-names>M</given-names> </name><name name-style="western"><surname>Luu</surname><given-names>AT</given-names> </name><name name-style="western"><surname>Bing</surname><given-names>L</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Chiruzzo</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ritter</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name></person-group><article-title>SeaExam and seabench: benchmarking llms with local multilingual questions in Southeast Asia</article-title><source>Findings of the Association for Computational Linguistics: NAACL 2025</source><year>2025</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>6134</fpage><lpage>6151</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.findings-naacl.341</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>P</given-names> </name><name name-style="western"><surname>Li</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>A</given-names> </name></person-group><article-title>Predicting dengue incidence in high-risk areas of China through the integration of Southeast Asian and local meteorological factors</article-title><source>Ecotoxicol Environ Saf</source><year>2025</year><month>01</month><day>15</day><volume>290</volume><fpage>117751</fpage><pub-id pub-id-type="doi">10.1016/j.ecoenv.2025.117751</pub-id><pub-id pub-id-type="medline">39837006</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Coker</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Hunter</surname><given-names>BM</given-names> </name><name name-style="western"><surname>Rudge</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Liverani</surname><given-names>M</given-names> </name><name name-style="western"><surname>Hanvoravongchai</surname><given-names>P</given-names> </name></person-group><article-title>Emerging infectious diseases in Southeast Asia: regional challenges to control</article-title><source>The Lancet</source><year>2011</year><month>02</month><volume>377</volume><issue>9765</issue><fpage>599</fpage><lpage>609</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(10)62004-1</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kleepbua</surname><given-names>J</given-names> </name><name name-style="western"><surname>Guan</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Early spatiotemporal patterns and population characteristics of the COVID-19 pandemic in Southeast Asia</article-title><source>Health Care (Don Mills)</source><year>2021</year><volume>9</volume><issue>9</issue><fpage>1220</fpage><pub-id pub-id-type="doi">10.3390/healthcare9091220</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Saba Villarroel</surname><given-names>PM</given-names> </name><name name-style="western"><surname>Gumpangseth</surname><given-names>N</given-names> </name><name name-style="western"><surname>Songhong</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Emerging and re-emerging zoonotic viral diseases in Southeast Asia: One Health challenge</article-title><source>Front Public Health</source><year>2023</year><volume>11</volume><fpage>1141483</fpage><pub-id pub-id-type="doi">10.3389/fpubh.2023.1141483</pub-id><pub-id pub-id-type="medline">37383270</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wakamiya</surname><given-names>S</given-names> </name><name name-style="western"><surname>Morita</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kano</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ohkuma</surname><given-names>T</given-names> </name><name name-style="western"><surname>Aramaki</surname><given-names>E</given-names> </name></person-group><article-title>Tweet classification toward Twitter-based disease surveillance: new data, methods, and evaluations</article-title><source>J Med Internet Res</source><year>2019</year><month>02</month><day>20</day><volume>21</volume><issue>2</issue><fpage>e12783</fpage><pub-id pub-id-type="doi">10.2196/12783</pub-id><pub-id pub-id-type="medline">30785407</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="web"><article-title>Introducing GPT-5</article-title><source>OpenAI</source><year>2026</year><access-date>2026-04-22</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://openai.com/index/introducing-gpt-5/">https://openai.com/index/introducing-gpt-5/</ext-link></comment></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sokolova</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lapalme</surname><given-names>G</given-names> </name></person-group><article-title>A systematic analysis of performance measures for classification tasks</article-title><source>Inf Process Manag</source><year>2009</year><month>07</month><volume>45</volume><issue>4</issue><fpage>427</fpage><lpage>437</lpage><pub-id pub-id-type="doi">10.1016/j.ipm.2009.03.002</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="report"><article-title>Ethical Guidelines for Medical and Biological Research Involving Human Subjects</article-title><year>2021</year><access-date>2026-09-12</access-date><publisher-name>Ministry of Education, Culture, Sports, Science and Technology (MEXT); Ministry of Health, Labour and Welfare (MHLW); Ministry of Economy, Trade and Industry (METI)</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.mext.go.jp/content/20250325-mxt_life-000035486-01.pdf">https://www.mext.go.jp/content/20250325-mxt_life-000035486-01.pdf</ext-link></comment></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Hwang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Roh</surname><given-names>HW</given-names> </name><name name-style="western"><surname>Park</surname><given-names>RW</given-names> </name></person-group><article-title>Performance of open-source large language models in psychiatry: usability study through comparative analysis of non-English records and English translations</article-title><source>J Med Internet Res</source><year>2025</year><month>08</month><day>18</day><volume>27</volume><fpage>e69857</fpage><pub-id pub-id-type="doi">10.2196/69857</pub-id><pub-id pub-id-type="medline">40825309</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ji</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Sia</surname><given-names>CH</given-names> </name><etal/></person-group><article-title>Large language model comparisons between English and Chinese query performance for cardiovascular prevention</article-title><source>Commun Med</source><year>2025</year><month>05</month><day>16</day><volume>5</volume><issue>1</issue><fpage>177</fpage><pub-id pub-id-type="doi">10.1038/s43856-025-00802-0</pub-id><pub-id pub-id-type="medline">40379850</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Mou</surname><given-names>W</given-names> </name><name name-style="western"><surname>Lai</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>P</given-names> </name></person-group><article-title>Language and cultural bias in AI: comparing the performance of large language models developed in different countries on Traditional Chinese Medicine highlights the need for localized models</article-title><source>J Transl Med</source><year>2024</year><month>03</month><day>29</day><volume>22</volume><issue>1</issue><fpage>319</fpage><pub-id pub-id-type="doi">10.1186/s12967-024-05128-4</pub-id><pub-id pub-id-type="medline">38553705</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Bagheri Nezhad</surname><given-names>S</given-names> </name><name name-style="western"><surname>Agrawal</surname><given-names>A</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Scherrer</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jauhiainen</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ljube&#x0161;i&#x0107;</surname><given-names>N</given-names> </name><name name-style="western"><surname>Zampieri</surname><given-names>M</given-names> </name><name name-style="western"><surname>Nakov</surname><given-names>P</given-names> </name><name name-style="western"><surname>Tiedemann</surname><given-names>J</given-names> </name></person-group><article-title>What drives performance in multilingual language models?</article-title><source>Proceedings of the Eleventh Workshop on NLP for Similar Languages, Varieties, and Dialects (VarDial 2024)</source><year>2024</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>16</fpage><lpage>27</lpage><pub-id pub-id-type="doi">10.18653/v1/2024.vardial-1.2</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Qin</surname><given-names>L</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>A survey of multilingual large language models</article-title><source>Patterns</source><year>2025</year><month>01</month><day>10</day><volume>6</volume><issue>1</issue><fpage>101118</fpage><pub-id pub-id-type="doi">10.1016/j.patter.2024.101118</pub-id><pub-id pub-id-type="medline">39896256</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eccles</surname><given-names>R</given-names> </name></person-group><article-title>Common cold</article-title><source>Front Allergy</source><year>2023</year><volume>4</volume><fpage>1224988</fpage><pub-id pub-id-type="doi">10.3389/falgy.2023.1224988</pub-id><pub-id pub-id-type="medline">37426629</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eccles</surname><given-names>R</given-names> </name></person-group><article-title>Understanding the symptoms of the common cold and influenza</article-title><source>Lancet Infect Dis</source><year>2005</year><month>11</month><volume>5</volume><issue>11</issue><fpage>718</fpage><lpage>725</lpage><pub-id pub-id-type="doi">10.1016/S1473-3099(05)70270-X</pub-id><pub-id pub-id-type="medline">16253889</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>D&#x2019;Amato</surname><given-names>G</given-names> </name><name name-style="western"><surname>Murrieta-Aguttes</surname><given-names>M</given-names> </name><name name-style="western"><surname>D&#x2019;Amato</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ansotegui</surname><given-names>IJ</given-names> </name></person-group><article-title>Pollen respiratory allergy: is it really seasonal?</article-title><source>World Allergy Organ J</source><year>2023</year><month>07</month><volume>16</volume><issue>7</issue><fpage>100799</fpage><pub-id pub-id-type="doi">10.1016/j.waojou.2023.100799</pub-id><pub-id pub-id-type="medline">37520612</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Juprasong</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sirirakphaisarn</surname><given-names>S</given-names> </name><name name-style="western"><surname>Siriwattanakul</surname><given-names>U</given-names> </name><name name-style="western"><surname>Songnuan</surname><given-names>W</given-names> </name></person-group><article-title>Exploring the effects of seasons, diurnal cycle, and heights on airborne pollen load in a Southeast Asian atmospheric condition</article-title><source>Front Public Health</source><year>2022</year><volume>10</volume><fpage>1067034</fpage><pub-id pub-id-type="doi">10.3389/fpubh.2022.1067034</pub-id><pub-id pub-id-type="medline">36589963</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singh</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Mathur</surname><given-names>C</given-names> </name></person-group><article-title>Climate change and pollen allergy in India and South Asia</article-title><source>Immunol Allergy Clin North Am</source><year>2021</year><month>02</month><volume>41</volume><issue>1</issue><fpage>33</fpage><lpage>52</lpage><pub-id pub-id-type="doi">10.1016/j.iac.2020.09.007</pub-id><pub-id pub-id-type="medline">33228871</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pham</surname><given-names>NT</given-names> </name><name name-style="western"><surname>Siddiquee</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sabit</surname><given-names>M</given-names> </name><name name-style="western"><surname>Grewling</surname><given-names>&#x0141;</given-names> </name></person-group><article-title>Monitoring, distribution and clinical relevance of airborne pollen and fern spores in Southeast Asia - a systematic review</article-title><source>World Allergy Organ J</source><year>2025</year><month>05</month><volume>18</volume><issue>5</issue><fpage>101053</fpage><pub-id pub-id-type="doi">10.1016/j.waojou.2025.101053</pub-id><pub-id pub-id-type="medline">40331224</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Savour&#x00E9;</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bousquet</surname><given-names>J</given-names> </name><name name-style="western"><surname>Jaakkola</surname><given-names>JJK</given-names> </name><name name-style="western"><surname>Jaakkola</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Jacquemin</surname><given-names>B</given-names> </name><name name-style="western"><surname>Nadif</surname><given-names>R</given-names> </name></person-group><article-title>Worldwide prevalence of rhinitis in adults: a review of definitions and temporal evolution</article-title><source>Clin Transl Allergy</source><year>2022</year><month>03</month><volume>12</volume><issue>3</issue><fpage>e12130</fpage><pub-id pub-id-type="doi">10.1002/clt2.12130</pub-id><pub-id pub-id-type="medline">35344304</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Katelaris</surname><given-names>CH</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>BW</given-names> </name><name name-style="western"><surname>Potter</surname><given-names>PC</given-names> </name><etal/></person-group><article-title>Prevalence and diversity of allergic rhinitis in regions of the world beyond Europe and North America</article-title><source>Clin Exp Allergy</source><year>2012</year><month>02</month><volume>42</volume><issue>2</issue><fpage>186</fpage><lpage>207</lpage><pub-id pub-id-type="doi">10.1111/j.1365-2222.2011.03891.x</pub-id><pub-id pub-id-type="medline">22092947</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Suchikova</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Tsybuliak</surname><given-names>N</given-names> </name><name name-style="western"><surname>Teixeira da Silva</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Nazarovets</surname><given-names>S</given-names> </name></person-group><article-title>GAIDeT (Generative AI Delegation Taxonomy): a taxonomy for humans to delegate tasks to generative artificial intelligence in scientific research and publishing</article-title><source>Account Res</source><year>2026</year><month>04</month><volume>33</volume><issue>3</issue><fpage>2544331</fpage><pub-id pub-id-type="doi">10.1080/08989621.2025.2544331</pub-id><pub-id pub-id-type="medline">40781729</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="web"><article-title>NTCIR-13 MedWeb [Article in Japanese]</article-title><source>NTCIR Project</source><access-date>2026-06-01</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://research.nii.ac.jp/ntcir/permission/ntcir-13/perm-ja-MedWeb.html">https://research.nii.ac.jp/ntcir/permission/ntcir-13/perm-ja-MedWeb.html</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Ablation studies.</p><media xlink:href="ai_v5i1e103321_app1.pdf" xlink:title="PDF File, 1030 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Error analysis.</p><media xlink:href="ai_v5i1e103321_app2.pdf" xlink:title="PDF File, 134 KB"/></supplementary-material></app-group></back></article>