<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Aging</journal-id><journal-id journal-id-type="publisher-id">aging</journal-id><journal-id journal-id-type="index">31</journal-id><journal-title>JMIR Aging</journal-title><abbrev-journal-title>JMIR Aging</abbrev-journal-title><issn pub-type="epub">2561-7605</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v9i1e93279</article-id><article-id pub-id-type="doi">10.2196/93279</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Multimodal Dementia Prediction With Large Language Models: Cross-Attention Over Text, Audio, and Image</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Agbavor</surname><given-names>Felix</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Liang</surname><given-names>Hualou</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib></contrib-group><aff id="aff1"><institution>School of Biomedical Engineering and Science, Drexel University</institution><addr-line>Philadelphia</addr-line><addr-line>PA</addr-line><country>United States</country></aff><aff id="aff2"><institution>Division of Artificial Intelligence and the Humanities, The Hong Kong Polytechnic University</institution><addr-line>HHB717, 7/F, 8 Hung Lok Road, Hung Hom</addr-line><addr-line>Kowloon</addr-line><country>China (Hong Kong)</country></aff><aff id="aff3"><institution>Department of Language Science and Technology, The Hong Kong Polytechnic University</institution><addr-line>Kowloon</addr-line><country>China (Hong Kong)</country></aff><aff id="aff4"><institution>Departments of Data Science and Artificial Intelligence, The Hong Kong Polytechnic University</institution><addr-line>Kowloon</addr-line><country>China (Hong Kong)</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>O'Connell</surname><given-names>Megan</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Prasetio</surname><given-names>Barlian Henryranu</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Fan</surname><given-names>Jin</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Hualou Liang, PhD, Division of Artificial Intelligence and the Humanities, The Hong Kong Polytechnic University, HHB717, 7/F, 8 Hung Lok Road, Hung Hom, Kowloon, China (Hong Kong), 852 2766-7697; <email>hualou.liang@polyu.edu.hk</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>21</day><month>9</month><year>2026</year></pub-date><volume>9</volume><elocation-id>e93279</elocation-id><history><date date-type="received"><day>10</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>07</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>31</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Felix Agbavor, Hualou Liang. Originally published in JMIR Aging (<ext-link ext-link-type="uri" xlink:href="https://aging.jmir.org">https://aging.jmir.org</ext-link>), 21.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Aging, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://aging.jmir.org">https://aging.jmir.org</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://aging.jmir.org/2026/1/e93279"/><abstract><sec><title>Background</title><p>Alzheimer disease (AD) is a leading cause of dementia, and there is growing interest in scalable approaches for early screening using speech-based tasks. While prior work has demonstrated promising results using either transcript-based language features or acoustic cues, most approaches remain unimodal or rely on simple fusion strategies that do not explicitly consider interactions across modalities.</p></sec><sec><title>Objective</title><p>In this study, we propose an attention-based trimodal fusion framework that integrates text, audio, and image representations of the Cookie Theft picture, which serves as the shared visual stimulus in the picture-description task.</p></sec><sec sec-type="methods"><title>Methods</title><p>Our method uses a new bidirectional cross-attention mechanism to achieve a unified multimodal embedding for downstream tasks. We evaluate the approach on 2 tasks: AD detection by classifying whether the participant has AD or not, and AD severity assessment by predicting Mini-Mental Status Examination cognitive scores.</p></sec><sec sec-type="results"><title>Results</title><p>On the AD detection task, trimodal fusion achieves the best overall performance (<italic>F</italic><sub>1</sub>-score=0.8667, area under the receiver operating characteristic curve=0.9032), outperforming unimodal baselines, bimodal fusion, and conventional early or late fusion methods. For AD severity assessment, the proposed multimodal representation reduces prediction error of root mean squared error to about 4.20, improving over both unimodal and bimodal fusion settings. We further perform the ablation analysis to show that bidirectional cross-attention consistently outperforms conventional unidirectional cross-attention.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>These results demonstrate that attention-based multimodal fusion can enhance dementia prediction from picture-description responses and provide a strong foundation for developing multimodal cognitive screening pipelines.</p></sec></abstract><kwd-group><kwd>Alzheimer disease</kwd><kwd>multimodal fusion</kwd><kwd>speech analysis</kwd><kwd>cognitive assessment</kwd><kwd>cross-attention</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Alzheimer disease (AD) is a progressive neurodegenerative disorder and a leading cause of dementia worldwide, with a growing public health impact due to aging populations [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Early screening and continuous monitoring are critical for timely intervention, planning, and patient support [<xref ref-type="bibr" rid="ref3">3</xref>]. However, conventional clinical assessments remain resource-intensive, often requiring specialized expertise, controlled administration environments, expensive clinical facilities, and repeated in-person visits [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref5">5</xref>]. These limitations motivate scalable, accessible, noninvasive, and low-burden approaches for cognitive screening that can be deployed in realistic settings.</p><p>Spontaneous speech has emerged as a particularly promising biomarker for dementia because it reflects a wide range of cognitive-linguistic processes, including lexical retrieval, syntactic organization, semantic coherence, and executive control [<xref ref-type="bibr" rid="ref6">6</xref>-<xref ref-type="bibr" rid="ref8">8</xref>]. A substantial body of work demonstrates that spontaneous speech contains clinically meaningful signals for AD screening and diagnosis. However, most existing studies remain unimodal, relying solely on either textual features derived from transcripts or acoustic features extracted from audio, thereby neglecting complementary information that is critical for robust inference. More recent approaches have explored bimodal models (eg, audio + text), which generally outperform unimodal baselines by aligning linguistic and acoustic cues [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref10">10</xref>]. Despite these advances, bimodal analyses primarily focus on local speech characteristics such as pauses and lexical diversity and often fail to capture global discourse-level phenomena, including disrupted topic maintenance and cross-modal inconsistencies between spoken narratives and visual image.</p><p>To address these limitations, we introduce image information into AD prediction by jointly modeling text, audio, and image representations in the Cookie Theft picture-description task. Because the Cookie Theft picture is identical across all participants, the visual modality is not intended to encode participant-specific variability; instead, it provides shared visual context that grounds the spoken narrative during multimodal fusion. This visual context helps relate the narrative to the scene being described and serves as a semantic anchor for fusion. Each modality contributes different information. Text captures lexical choice, semantic content, coherence, and informativeness of the spoken description. Audio contributes complementary paralinguistic cues, including pauses, hesitation, rhythm, fluency, and other prosodic features not fully preserved in transcripts. The Cookie Theft picture provides shared visual context that grounds the narrative and helps assess how well the spoken description aligns with the scene.</p><p>Multimodal integration is commonly approached through 3 fusion strategies: early (data-level), intermediate (joint), and late (decision-level) fusion. Early fusion projects features from different modalities into a single shared space, but it often fails to capture higher-order cross-modal interactions. Late fusion, by contrast, aggregates the outputs of modality-specific models, preserving individual modality strengths while overlooking deeper interdependencies among them [<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref13">13</xref>]. Intermediate, or joint, fusion offers a principled compromise by explicitly modeling interactions across modalities during representation learning. In this context, Transformer-based cross-attention has emerged as a key mechanism for capturing fine-grained and long-range intermodal relationships [<xref ref-type="bibr" rid="ref14">14</xref>]. Intuitively, cross-attention allows one modality to look at another and selectively use the parts that are most relevant for the task. For example, a text representation can attend to complementary information in the audio or visual representation, so the fused embedding reflects not only what was said, but also how it was spoken and how well it aligns with the Cookie Theft picture context.</p><p>Cross-attention has been widely adopted in language-vision representation learning [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref15">15</xref>], audio-text integration [<xref ref-type="bibr" rid="ref11">11</xref>], and more recently in dementia prediction [<xref ref-type="bibr" rid="ref16">16</xref>]. However, existing approaches typically rely on unidirectional cross-attention, in which one modality queries another, resulting in asymmetric information flow. To overcome this limitation, we propose a bidirectional cross-attention mechanism that enables symmetric and iterative information exchange across modalities. In this design, each modality both attends to and is updated by the other, allowing representations to be refined in both directions. This iterative bidirectional interaction facilitates richer cross-modal integration [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref15">15</xref>], promotes balanced modality contributions, and mitigates the risk of modality dominance. Building on this perspective, our trimodal fusion architecture leverages bidirectional cross-attention to produce unified embeddings that capture both complementary evidence and cross-modal agreement, rather than relying on fixed or heuristic fusion rules. Simply put, rather than allowing only one modality to query information from another, bidirectional cross-attention enables both modalities to interact and refine each other, resulting in a fused representation derived from 2-way information exchange rather than 1-way conditioning.</p><p>Taken together, we propose a trimodal embedding-level fusion architecture that integrates text, audio, and image through bidirectional cross-attention (<xref ref-type="fig" rid="figure1">Figure 1</xref>). The resulting fused embedding is used for downstream AD detection and cognitive score prediction. We evaluate this approach on the ADReSSo 2021 picture-description task against unimodal baselines, bimodal fusion models, and conventional early- and late-fusion strategies.</p><p>Our main contributions are as follows:</p><list list-type="bullet"><list-item><p>We introduce a staged bidirectional cross-attention fusion framework for integrating text, audio, and image representations in the AD picture-description setting.</p></list-item><list-item><p>On the ADReSSo 2021 picture-description task, we show that trimodal fusion improves over unimodal baselines and outperforms bimodal fusion for both AD detection and Mini-Mental Status Examination (MMSE) prediction.</p></list-item><list-item><p>We show that the proposed fusion framework performs better than conventional fusion strategies, including early fusion and late fusion.</p></list-item><list-item><p>We further show that bidirectional cross-attention yields more effective and stable performance than a standard 1-way cross-attention under the same task setting.</p></list-item></list><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Proposed trimodal fusion architecture for dementia prediction. Pretrained 768-dimensional embeddings are extracted from text (ModernBERT), audio (Wav2Vec 2.0), and the Cookie Theft picture (contrastive language&#x2013;image pretraining [CLIP] ViT-L/14). Fusion is performed in 2 stages for clarity. In stage 1, text and image are fused through a bidirectional bimodal cross-attention block to obtain an intermediate representation, <italic>TI</italic>. In stage 2, this intermediate <italic>TI</italic> representation is fused with audio through the same bidirectional fusion block to produce the final multimodal embedding, <italic>z</italic>, which is used for downstream classification and regression. The inset illustrates the generic bimodal fusion module used in each stage, where 2 modality embeddings are updated sequentially across layers by attending to one another, enabling reciprocal information exchange beyond standard 1-way cross-attention.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="aging_v9i1e93279_fig01.png"/></fig></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Dataset Description</title><p>The dataset used in this study is derived from the ADReSSo 2021 Challenge and consists of speech recordings from a picture-description task [<xref ref-type="bibr" rid="ref8">8</xref>], where cognitively normal participants and individuals diagnosed with AD were asked to describe the Cookie Theft image from the Boston Diagnostic Aphasia Examination [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. In total, the dataset contains 237 recordings, split into a 70/30 train&#x2013;test partition with demographic balancing, resulting in 166 samples for training and 71 for testing. Within the training set, there are 87 AD and 79 non-AD (healthy control) recordings.</p><p>A key strength of ADReSSo is its focus on reducing common sources of bias in dementia screening benchmarks. The dataset was carefully curated to mitigate confounds such as repeated speech samples from the same individual, variation in recording quality, and imbalances in age and gender. Demographic matching was performed using a propensity-score procedure as described by Luz et al [<xref ref-type="bibr" rid="ref8">8</xref>], yielding standardized mean differences below 0.001 for both age and gender, which helps ensure that downstream model performance reflects disease-related signal rather than demographic artifacts.</p></sec><sec id="s2-2"><title>Dataset Preprocessing and Feature Representation</title><p>We preprocess the ADReSSo picture-description recordings to obtain aligned audio, text, and image representations, which serve as inputs to our multimodal fusion architecture. For the audio modality, we extract embeddings using Wav2Vec 2.0 [<xref ref-type="bibr" rid="ref19">19</xref>], a self-supervised speech representation model trained to learn rich acoustic features directly from raw waveforms. Given an input recording, we obtain a sequence of frame-level hidden states and apply mean pooling over time to produce a fixed-dimensional 768-D audio embedding per sample.</p><p>For the text modality, we first transcribe each recording using OpenAI Whisper [<xref ref-type="bibr" rid="ref20">20</xref>], which provides robust automatic speech recognition for spontaneous speech. To better preserve clinically relevant phenomena commonly observed in AD speech, we supply an initial decoding prompt (<italic>&#x201C;</italic>Umm, Uhh, let me think like, hmm... Okay, here&#x2019;s what I&#x2019;m, like, thinking<italic>."</italic>) that encourages retention of fillers, disfluencies, repetitions, and incomplete phrases, rather than normalizing them away. We then encode the resulting transcripts using ModernBERT [<xref ref-type="bibr" rid="ref21">21</xref>], a modern bidirectional Transformer encoder optimized for producing strong sentence-level representations. Similar to the audio pipeline, we derive a fixed-length 768-D text embedding for each transcript, providing a compact semantic representation that captures content, coherence, and lexical organization.</p><p>For the image modality, we represent the Cookie Theft stimulus using contrastive language&#x2013;image pretraining (CLIP; ViT-L/14), a vision-language model trained via contrastive learning to align visual and textual representations in a shared embedding space. We extract the image encoder output and use a 768-D image embedding, which provides a high-level description of the visual scene structure and salient objects.</p></sec><sec id="s2-3"><title>Attention-Based Multimodal Fusion</title><p>A central component of our fusion approach is bidirectional cross-attention, where each modality is updated by attending to the others, allowing text, audio, and image embeddings to exchange information in both directions rather than being combined through simple concatenation. Unlike simple fusion methods (eg, concatenation), cross-attention allows the model to learn which parts of one representation are most relevant given the other representation, which is especially important in dementia screening where different modalities may carry complementary cues. Formally, we use the standard attention operation represented by <inline-formula><mml:math id="ieqn1"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mrow><mml:mi mathvariant="normal">A</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mtext>=</mml:mtext><mml:mrow><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">f</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">x</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mi>Q</mml:mi><mml:msup><mml:mi>K</mml:mi><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:msqrt><mml:msub><mml:mi>d</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:msqrt></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mi>V</mml:mi></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula>, where <italic>Q</italic> (queries) represents what we want to &#x201C;look for,&#x201D; while <italic>K</italic> (keys) and <italic>V</italic> (values) represent the information we want to retrieve from.</p><p>Intuitively, bidirectional cross-attention can be understood as a 2-way exchange between modalities. In one direction, the text representation asks which parts of the audio or image are most relevant for interpreting what was said; in the other direction, the audio or image representation is also updated by attending back to the text. This allows the fused representation to reflect not only the content of the spoken description, but also how it was delivered and how well it aligns with the shared Cookie Theft picture context. In practice, the bidirectional design helps the model avoid treating one modality as fixed context for another and instead encourages reciprocal refinement of the modality embeddings across layers.</p><p>To combine information from text, image, and audio in a way that goes beyond simple concatenation, we propose an attention-based fusion module (<xref ref-type="fig" rid="figure1">Figure 1</xref>) that learns how modalities complement each other when predicting AD status and cognitive outcomes. The key idea is that each modality carries different but related information: text reflects linguistic content and coherence, audio captures acoustic patterns such as hesitation and fluency, and the image provides the visual context that grounds the picture-description narrative. Rather than assuming each modality contributes equally, our fusion approach learns to emphasize the most informative signals dynamically.</p><p>Our architecture performs multimodal fusion using bidirectional cross-attention, enabling each modality to iteratively incorporate information from the others. Rather than simply concatenating embeddings, the model allows the text representation to attend to complementary cues from the image and audio, while the non-text modalities also attend back to the evolving text representation. This bidirectional exchange supports richer alignment between what is said, how it is spoken, and the visual context of the picture-description stimulus. As a result, the fusion module produces a single multimodal embedding that reflects both modality-specific evidence and cross-modal dependencies, which is then used for downstream AD classification and its severity assessment.</p><p>For the implementation, we use a multilayer attention block that alternates between the 2 inputs, allowing one modality to &#x201C;look at&#x201D; and incorporate information from the other. Intuitively, this helps the model learn cross-modal interactions. For example, how acoustic hesitation patterns relate to lexical choice or how a visually grounded narrative differs between cognitively normal and impaired speakers. The final output is a single fused embedding per sample, which is passed to downstream classifiers or regressors for AD detection and its severity assessment.</p></sec><sec id="s2-4"><title>Training Regimen</title><p>We train the proposed trimodal fusion model using the ADReSSo training split and further reserve 20% of the training data as a validation set for model selection and early stopping. This validation split is created randomly but reproducibly using a fixed seed of 42, ensuring consistent comparisons across experiments. To enforce reproducibility, the same seed is applied across Python, NumPy, and PyTorch (central processing unit and Compute Unified Device Architecture), with deterministic CUDA deep neural network behavior enabled. During training, samples are loaded in batches of 32. The fusion backbone uses a shared projection dimension of 768, with 4 attention heads and 3 cross-attention layer pairs per fusion stage, and models are trained for up to 50 epochs with a dropout rate of 0.1. To reduce scale mismatch across modalities, we apply L2 normalization to the input embeddings prior to fusion.</p><p>The model parameters are optimized using AdamW with a learning rate of 0.0001 and weight decay of 1 &#x00D7; 10<sup>&#x2212;4</sup>. For AD detection, we minimize cross-entropy loss, and for AD severity assessment, we use mean squared error loss. To stabilize learning, we use a ReduceLROnPlateau scheduler that monitors validation loss and reduces the learning rate by a factor of 0.5 after 2 consecutive epochs without improvement. We additionally apply gradient clipping with a maximum norm of 1.0. Early stopping is based on validation loss with a patience of 5 epochs, and the model weights from the epoch with the lowest validation loss are restored for final evaluation and embedding extraction.</p></sec><sec id="s2-5"><title>AD Detection Task</title><p>The AD detection task is formulated as a binary classification problem, where the goal is to distinguish between participants with AD and cognitively normal (non-AD) participants using picture-description responses from the ADReSSo dataset. Because dementia-related signals can manifest across multiple modalities, we evaluate models using (1) audio-only representations, (2) text-only representations, and (3) multimodal fusion representations derived from combinations of text, audio, and image. Audio representations are extracted from Wav2Vec 2.0 embeddings, text representations are extracted from ModernBERT embeddings computed over Whisper transcripts, and image representations are extracted using CLIP ViT-L/14. We primarily use 3 standard machine-learning models for the AD classification task. These include support vector classifier (SVC), random forest (RF), and logistic regression (LR), all of which are implemented from the scikit-learn package [<xref ref-type="bibr" rid="ref22">22</xref>-<xref ref-type="bibr" rid="ref25">25</xref>].</p></sec><sec id="s2-6"><title>AD Severity Assessment Task</title><p>In addition to AD classification, we evaluate our approach for AD severity assessment by predicting MMSE cognitive score, formulated as a regression problem. The objective is to predict a participant&#x2019;s cognitive score directly from their picture-description response, enabling a finer-grained estimate of cognitive status beyond binary diagnosis. Similar to the classification setting, we investigate the predictive use of multiple modalities by training models with audio-only, text-only, and image-only representations, as well as multimodal fusion representations that combine complementary cues across modalities. Specifically, we use support vector regressor (SVR), random forest regressor (RFR), and ridge regression (Ridge) as our machine learning models [<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref26">26</xref>]. For this task, we use the same feature representations described in the preprocessing pipeline: Wav2Vec 2.0 embeddings for audio, ModernBERT embeddings derived from Whisper transcripts for text, and CLIP ViT-L/14 embeddings for the image stimulus. We compare our proposed attention-based trimodal fusion model against unimodal baselines and bimodal fusion models.</p></sec><sec id="s2-7"><title>Evaluation Metrics</title><p>We evaluate performance separately for the AD classification and AD severity assessment tasks. For the binary classification setting, we report accuracy, precision, recall, and <italic>F</italic><sub>1</sub>-score, providing a balanced view of overall correctness as well as performance on the positive AD class. Additionally, we also show the receiver operating characteristic (ROC) curve for modality comparisons and report the area under the ROC curve. For the regression task, we quantify prediction error using root mean squared error (RMSE), which measures the average magnitude of score prediction deviations. For all classical machine-learning baselines, optimal hyperparameters are selected using grid search with 5-fold cross-validation performed on the training split only. This ensures that model configuration is tuned in a statistically robust manner while keeping the held-out test set strictly reserved for final evaluation.</p><p>To better characterize uncertainty in model performance, we additionally report 95% CIs for the major classification and regression results using nonparametric bootstrap resampling of the held-out test set. For each metric, CIs were estimated from the empirical bootstrap distribution across 1000 repeated resamples. Note that the observed performance differences were not tested for statistical significance to avoid overreliance on arbitrary significance thresholds (eg, <italic>P</italic>&#x003C;.05) and potential misinterpretations of <italic>P</italic> values.</p></sec><sec id="s2-8"><title>Attention Variant Study (Bidirectional Cross-Attention Versus Cross-Attention Only)</title><p>To assess the impact of reciprocal information exchange during multimodal fusion, we compare 2 attention configurations under an otherwise identical fusion framework. The cross-attention-only, namely standard cross-attention, performs fusion using a single cross-attention direction, where 1 modality attends to another to incorporate complementary information. In contrast, the bidirectional cross-attention applies cross-attention in both directions, allowing each modality to be updated using information from the other in a sequential manner. Both variants use the same pretrained embeddings, projection dimensionality, optimization settings, and train or validation split, enabling a controlled comparison of the attention mechanisms.</p></sec><sec id="s2-9"><title>Comparison with Conventional Fusion Methods</title><p>To assess whether the proposed attention-based multimodal fusion provides benefits beyond standard approaches, we compare it against 2 widely used conventional fusion strategies for the AD classification task: early fusion and late fusion. These baselines are commonly adopted in multimodal machine learning due to their simplicity, but they often struggle to capture complex interactions between modalities. For early fusion, we construct a single feature vector by concatenating the modality-specific embeddings (text, audio, and image) and train a classifier directly on the combined representation. While straightforward, early fusion does not explicitly model cross-modal relationships and instead relies on the downstream classifier to learn interactions implicitly from the concatenated embedding.</p><p>For late fusion, we train separate unimodal classifiers and combine their predictions at the decision level using majority voting. Because the visual stimulus is identical across all samples in the ADReSSo picture-description setting, an image-only classifier would not be meaningful for late fusion. Therefore, we report late-fusion performance using only the text and audio unimodal models. This comparison helps isolate the benefit of our attention-based fusion, which can jointly reason over modalities and learn cross-modal dependencies rather than combining independent unimodal decisions.</p><p>In addition to these conventional fusion strategies, we include a neural network baseline trained on the concatenated multimodal embeddings. The neural network architecture consisted of 2 hidden layers with 512 and 128 nodes, respectively, using the rectified linear unit activation function. This funnel-like bottleneck design by progressively stepping down from the high-dimensional input space to 512 and then 128 nodes was chosen to systematically compress the multimodal features. This encourages the network to learn robust, high-level representations while mitigating the risk of overfitting on the relatively small dataset [<xref ref-type="bibr" rid="ref27">27</xref>]. The network was optimized using the Adam solver with an initial learning rate of 0.001. To further mitigate overfitting and ensure optimal generalization, we implemented early stopping. Specifically, 10% of the training data were set aside as a validation set, and training was terminated if the validation score did not improve by at least 1 &#x00D7; 10<sup>&#x2212;4</sup> for 10 consecutive epochs. This baseline provides a stronger learned comparison than linear or tree-based classifiers on early-fusion features, while still operating on the same underlying embedding inputs. In this way, we can assess whether any observed gains arise simply from using a more flexible classifier on concatenated features or from the proposed bidirectional fusion mechanism itself.</p></sec><sec id="s2-10"><title>Ethical Considerations</title><p>The ADReSSo Challenge data used in this study are available through DementiaBank with approved credentials. The studies involving human participants were reviewed and approved by the DementiaBank consortium. All enrolled participants provided informed written consent to participate in this study. All data analyses in this work were conducted using deidentified data.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>AD Detection Task</title><sec id="s3-1-1"><title>Unimodal Baselines</title><p>In <xref ref-type="table" rid="table1">Table 1</xref>, we compare the performance of text and audio embeddings using the different classifiers. From the table, we can see that text provides stronger prediction for AD than the audio. With ModernBERT embeddings, SVC attains an <italic>F</italic><sub>1</sub>-score of 0.8334, outperforming LR which has an <italic>F</italic><sub>1</sub>-score of 0.7910 and RF which has an <italic>F</italic><sub>1</sub>-score of 0.7713. In general, audio (wav2vec2) is weaker across all the classifiers as the best audio-only result is achieved by RF with an <italic>F</italic><sub>1</sub>-score of 0.6736 and an accuracy of 0.7048. This performance gap between text and audio underscores that picture-description transcripts capture more discriminative information than the audio embeddings.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Unimodal baselines on the ADReSSo 2021 unseen test set (n=71)<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Classifier</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom">Precision (95% CI)</td><td align="left" valign="bottom">Recall (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">SVC<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup><sub>text</sub></td><td align="left" valign="top">0.8448 (0.7606&#x2010;0.9296)</td><td align="left" valign="top">0.8744 (0.7500&#x2010;0.9714)</td><td align="left" valign="top">0.7999 (0.6571&#x2010;0.9231)</td><td align="left" valign="top">0.8334 (0.7241&#x2010;0.9231)</td></tr><tr><td align="left" valign="top">SVC<sub>audio</sub></td><td align="left" valign="top">0.7051 (0.5915&#x2010;0.8028)</td><td align="left" valign="top">0.7069 (0.5484&#x2010;0.8529)</td><td align="left" valign="top">0.6852 (0.5172&#x2010;0.8294)</td><td align="left" valign="top">0.6928 (0.5588&#x2010;0.8095)</td></tr><tr><td align="left" valign="top">LR<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup><sub>text</sub></td><td align="left" valign="top">0.7876 (0.6901&#x2010;0.8732)</td><td align="left" valign="top">0.7618 (0.6190&#x2010;0.8919)</td><td align="left" valign="top">0.8275 (0.6969&#x2010;0.9429)</td><td align="left" valign="top">0.7910 (0.6774&#x2010;0.8889)</td></tr><tr><td align="left" valign="top">LR<sub>audio</sub></td><td align="left" valign="top">0.6617 (0.5493&#x2010;0.7746)</td><td align="left" valign="top">0.6886 (0.5161&#x2010;0.8519)</td><td align="left" valign="top">0.5712 (0.3953&#x2010;0.7317)</td><td align="left" valign="top">0.6207 (0.4666&#x2010;0.7500)</td></tr><tr><td align="left" valign="top">RF<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup><sub>text</sub></td><td align="left" valign="top">0.8029 (0.7042&#x2010;0.8873)</td><td align="left" valign="top">0.8886 (0.7500&#x2010;1.0000)</td><td align="left" valign="top">0.6859 (0.5294&#x2010;0.8333)</td><td align="left" valign="top">0.7713 (0.6429&#x2010;0.8772)</td></tr><tr><td align="left" valign="top">RF<sub>audio</sub></td><td align="left" valign="top">0.7048 (0.5915&#x2010;0.8169)</td><td align="left" valign="top">0.7338 (0.5714&#x2010;0.8846)</td><td align="left" valign="top">0.6284 (0.4643&#x2010;0.7858)</td><td align="left" valign="top">0.6736 (0.5283&#x2010;0.7949)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Performance of text-only (ModernBERT embeddings) and audio-only (wav2vec 2.0 embeddings) representations with 3 standard classifiers (support vector classifier, logistic regression, and random forest) is shown. </p></fn><fn id="table1fn2"><p><sup>b</sup>SVC: support vector classifier.</p></fn><fn id="table1fn3"><p><sup>c</sup>LR: logistic regression.</p></fn><fn id="table1fn4"><p><sup>d</sup>RF: random forest.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-1-2"><title>Bimodal Fusion (2-Way)</title><p>Combining text and image embeddings yields consistently strong performance as shown in <xref ref-type="table" rid="table2">Table 2</xref>, reflecting the benefit of pairing transcript semantics with visual context from the Cookie Theft stimulus. Across classifiers, the results remain close to the text-only baselines, suggesting that the image modality provides incremental gain rather than a major boost in this setting. The top configuration is achieved by SVC, reaching an <italic>F</italic><sub>1</sub>-score of 0.8371 with an accuracy of 0.8437, slightly exceeding the text-only SVC <italic>F</italic><sub>1</sub>-score (0.8358). RF also performs competitively (<italic>F</italic><sub>1</sub>-score=0.8428), while LR attains an <italic>F</italic><sub>1</sub>-score of 0.8011. Overall, text continues to dominate prediction quality, while image representations appear to offer modest complementary signal.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Text + image fusion for Alzheimer disease (AD) classification<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Classifier</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom">Precision (95% CI)</td><td align="left" valign="bottom">Recall (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">SVC<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="top">0.8437 (0.7465&#x2010;0.9296)</td><td align="left" valign="top">0.8509 (0.7241&#x2010;0.9655)</td><td align="left" valign="top">0.8275 (0.6944&#x2010;0.9429)</td><td align="left" valign="top">0.8371 (0.7333&#x2010;0.9247)</td></tr><tr><td align="left" valign="top">LR<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">0.8011 (0.7042&#x2010;0.8873)</td><td align="left" valign="top">0.7978 (0.6571&#x2010;0.9231)</td><td align="left" valign="top">0.7987 (0.6571&#x2010;0.9259)</td><td align="left" valign="top">0.7959 (0.6857&#x2010;0.8919)</td></tr><tr><td align="left" valign="top">RF<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">0.8428 (0.7606&#x2010;0.9296)</td><td align="left" valign="top">0.8731 (0.7500&#x2010;0.9714)</td><td align="left" valign="top">0.7987 (0.6571&#x2010;0.9286)</td><td align="left" valign="top">0.8321 (0.7241&#x2010;0.9231)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Performance of bimodal text + image representations on the ADReSSo 2021 unseen test set (n=71) using 3 classifiers (support vector classifier, logistic regression, and random forest) is shown. Text embeddings are extracted with ModernBERT, and image embeddings are extracted with contrastive language&#x2013;image pretraining ViT-L/14. </p></fn><fn id="table2fn2"><p><sup>b</sup>SVC: support vector classifier.</p></fn><fn id="table2fn3"><p><sup>c</sup>LR: logistic regression.</p></fn><fn id="table2fn4"><p><sup>d</sup>RF: random forest.</p></fn></table-wrap-foot></table-wrap><p>The text + audio pairing produces the best bimodal performance in <xref ref-type="table" rid="table3">Table 3</xref> among all 2-way combinations. In particular, incorporating audio improves performance beyond transcript-only models, suggesting that features such as speech rhythm, hesitation patterns, and fluency irregularities provide additional discriminative value. The best overall results are obtained with SVC, which achieves an accuracy of 0.8584 and an <italic>F</italic><sub>1</sub>-score of 0.8503, outperforming both the text-only SVC model (<italic>F</italic><sub>1</sub>-score=0.8358) and the other bimodal settings. RF remains strong (<italic>F</italic><sub>1</sub>-score=0.8325), whereas LR performs slightly lower (<italic>F</italic><sub>1</sub>-score=0.7903). These results support the benefit of integrating acoustic information when strong text representations are available.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Text + audio fusion for Alzheimer disease (AD) classification<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Classifier</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom">Precision (95% CI)</td><td align="left" valign="bottom">Recall (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">SVC<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="top">0.8584 (0.7746&#x2010;0.9296)</td><td align="left" valign="top">0.8771 (0.7568&#x2010;0.9722)</td><td align="left" valign="top">0.8286 (0.6944&#x2010;0.9444)</td><td align="left" valign="top">0.8503 (0.7458&#x2010;0.9315)</td></tr><tr><td align="left" valign="top">LR<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">0.8016 (0.7042&#x2010;0.8873)</td><td align="left" valign="top">0.8165 (0.6765&#x2010;0.9394)</td><td align="left" valign="top">0.7704 (0.6249&#x2010;0.9024)</td><td align="left" valign="top">0.7903 (0.6762&#x2010;0.8889)</td></tr><tr><td align="left" valign="top">RF<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">0.8440 (0.7606&#x2010;0.9296)</td><td align="left" valign="top">0.8732 (0.7500&#x2010;0.9714)</td><td align="left" valign="top">0.7993 (0.6571&#x2010;0.9231)</td><td align="left" valign="top">0.8325 (0.7213&#x2010;0.9206)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Performance of bimodal text + audio representations on the ADReSSo 2021 unseen test set (n=71) using 3 classifiers (support vector classifier, logistic regression, and random forest) is shown. Text embeddings are extracted with ModernBERT, and audio embeddings are extracted with Wav2Vec 2.0. </p></fn><fn id="table3fn2"><p><sup>b</sup>SVC: support vector classifier.</p></fn><fn id="table3fn3"><p><sup>c</sup>LR: logistic regression.</p></fn><fn id="table3fn4"><p><sup>d</sup>RF: random forest.</p></fn></table-wrap-foot></table-wrap><p><xref ref-type="table" rid="table4">Table 4</xref> shows results for audio + image fusion, a pairing that excludes transcript information and relies only on acoustic patterns and visual context. Performance is notably lower than text-based bimodal models, but it improves substantially over audio-only baselines, suggesting that image features contribute useful grounding signals when linguistic embeddings are not available. The top performance is achieved by SVC, with an accuracy of 0.7188 and an <italic>F</italic><sub>1</sub>-score of 0.6934, followed closely by RF (<italic>F</italic><sub>1</sub>-score=0.6830). LR produces a slightly weaker result (<italic>F</italic><sub>1</sub>-score=0.6530). Overall, these results reinforce that while audio + image alone can support AD prediction above chance, transcript-derived text embeddings remain critical for achieving top performance on picture-description based dementia screening.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Audio + image fusion for Alzheimer disease (AD) classification<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Classifier</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom">Precision (95% CI)</td><td align="left" valign="bottom">Recall (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">SVC<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">0.7188 (0.6197&#x2010;0.8169)</td><td align="left" valign="top">0.7427 (0.5806&#x2010;0.8889)</td><td align="left" valign="top">0.6558 (0.4997&#x2010;0.8065)</td><td align="left" valign="top">0.6934 (0.5574&#x2010;0.8116)</td></tr><tr><td align="left" valign="top">LR<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">0.6760 (0.5634&#x2010;0.7746)</td><td align="left" valign="top">0.6869 (0.5185&#x2010;0.8438)</td><td align="left" valign="top">0.6283 (0.4643&#x2010;0.7857)</td><td align="left" valign="top">0.6530 (0.5098&#x2010;0.7742)</td></tr><tr><td align="left" valign="top">RF<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top">0.7180 (0.6056&#x2010;0.8169)</td><td align="left" valign="top">0.7582 (0.6000&#x2010;0.9091)</td><td align="left" valign="top">0.6270 (0.4571&#x2010;0.7812)</td><td align="left" valign="top">0.6830 (0.5385&#x2010;0.8056)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Performance of bimodal audio + image representations on the ADReSSo 2021 unseen test set (n=71) using 3 classifiers (support vector classifier, logistic regression, and random forest) is shown. Audio embeddings are extracted with Wav2Vec 2.0, and image embeddings are extracted with CLIP ViT-L/14. </p></fn><fn id="table4fn2"><p><sup>b</sup>SVC: support vector classifier.</p></fn><fn id="table4fn3"><p><sup>c</sup>LR: logistic regression.</p></fn><fn id="table4fn4"><p><sup>d</sup>RF: random forest.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-1-3"><title>Multimodal Fusion (3-Way)</title><p>Using all 3 modalities (text, audio, and image) leads to the highest AD classification performance in our experiments. Among the 3 classifiers, SVC achieves the best overall results, with an accuracy of 0.8722 and an <italic>F</italic><sub>1</sub>-score of 0.8667. LR attains an <italic>F</italic><sub>1</sub>-score of 0.8542, while RF achieves an <italic>F</italic><sub>1</sub>-score of 0.8371. Overall, the 3-way fusion setting produces strong and stable performance across classifiers, with SVC providing the highest classification scores (<xref ref-type="table" rid="table5">Table 5</xref>).</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Multimodal fusion for Alzheimer disease (AD) classification<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Classifier</td><td align="left" valign="bottom">Accuracy (95% CI)</td><td align="left" valign="bottom">Precision (95% CI)</td><td align="left" valign="bottom">Recall (95% CI)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top">SVC<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup></td><td align="left" valign="top">0.8722 (0.7887&#x2010;0.9437)</td><td align="left" valign="top">0.8806 (0.7632&#x2010;0.9730)</td><td align="left" valign="top">0.8564 (0.7317&#x2010;0.9688)</td><td align="left" valign="top">0.8667 (0.7692&#x2010;0.9474)</td></tr><tr><td align="left" valign="top">LR<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup></td><td align="left" valign="top">0.8582 (0.7746&#x2010;0.9437)</td><td align="left" valign="top">0.8558 (0.7250&#x2010;0.9677)</td><td align="left" valign="top">0.8563 (0.7273&#x2010;0.9678)</td><td align="left" valign="top">0.8542 (0.7536&#x2010;0.9367)</td></tr><tr><td align="left" valign="top">RF<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td><td align="left" valign="top">0.8437 (0.7465&#x2010;0.9296)</td><td align="left" valign="top">0.8509 (0.7241&#x2010;0.9655)</td><td align="left" valign="top">0.8275 (0.6944&#x2010;0.9429)</td><td align="left" valign="top">0.8371 (0.7333&#x2010;0.9247)</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>Performance of text + audio + image fusion on the ADReSSo 2021 unseen test set (n=71) using 3 classifiers (support vector classifier, logistic regression, random forest). Text embeddings are extracted with ModernBERT, audio embeddings with Wav2Vec 2.0, and image embeddings with CLIP ViT-L/14.</p></fn><fn id="table5fn2"><p><sup>b</sup>SVC: support vector classifier.</p></fn><fn id="table5fn3"><p><sup>c</sup>LR: logistic regression.</p></fn><fn id="table5fn4"><p><sup>d</sup>RF: random forest.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s3-2"><title>ROC Curve Comparison Between Bimodal and Multimodal Fusion</title><p><xref ref-type="fig" rid="figure2">Figure 2</xref> compares ROC curves across bimodal fusion settings and the proposed trimodal multimodal fusion model for AD classification. Among the bimodal approaches, text + audio achieves the highest discrimination performance (area under the receiver operating characteristic curve [AUC]=0.8714), followed closely by text + image (AUC=0.8595), while audio + image performs notably lower (AUC=0.7937). The proposed multimodal fusion model yields the strong overall ROC performance with an AUC of 0.9032, indicating improved separability between participants with AD and without AD relative to all bimodal alternatives.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Receiver operating characteristic (ROC) comparison across bimodal and multimodal fusion (Alzheimer disease [AD] classification). ROC curves on the ADReSSo 2021 unseen test set (n=71) compare bimodal fusion models (audio + image, text + image, text + audio) against the proposed trimodal multimodal fusion (text + audio + image). The legend reports the area under the receiver operating characteristic curve (AUC) for each method.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="aging_v9i1e93279_fig02.png"/></fig></sec><sec id="s3-3"><title>Comparison to Other Fusion Methods</title><p><xref ref-type="table" rid="table6">Table 6</xref> compares the proposed multimodal fusion with conventional early fusion, late fusion, and a neural network baseline trained on concatenated embeddings. For SVC, both early fusion and late fusion achieve an accuracy of 0.8169, with <italic>F</italic><sub>1</sub>-scores of 0.7937 and 0.8000, respectively. The neural network baseline improves this comparison to an accuracy of 0.8311 and an <italic>F</italic><sub>1</sub>-score of 0.8099. In contrast, the proposed multimodal fusion achieves a higher accuracy of 0.8722 and a higher <italic>F</italic><sub>1</sub>-score of 0.8667. A similar pattern is observed across the other classifier families, indicating that the advantage of the proposed fusion is not limited to comparison against only simple conventional baselines.</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Comparison with conventional and learned fusion baselines for Alzheimer disease (AD) classification<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup>.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Fusion method</td><td align="left" valign="bottom">Accuracy</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top">SVC<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup><sub>early fusion</sub></td><td align="left" valign="top">0.8169</td><td align="left" valign="top">0.8929</td><td align="left" valign="top">0.7143</td><td align="left" valign="top">0.7937</td></tr><tr><td align="left" valign="top">SVC<sub>late fusion</sub></td><td align="left" valign="top">0.8169</td><td align="left" valign="top">0.8666</td><td align="left" valign="top">0.7428</td><td align="left" valign="top">0.8000</td></tr><tr><td align="left" valign="top">SVC<sub>multimodal fusion</sub></td><td align="left" valign="top">0.8722</td><td align="left" valign="top">0.8806</td><td align="left" valign="top">0.8564</td><td align="left" valign="top">0.8667</td></tr><tr><td align="left" valign="top">LR<sup><xref ref-type="table-fn" rid="table6fn3">c</xref></sup><sub>early fusion</sub></td><td align="left" valign="top">0.8028</td><td align="left" valign="top">0.8387</td><td align="left" valign="top">0.7429</td><td align="left" valign="top">0.7429</td></tr><tr><td align="left" valign="top">LR<sub>late fusion</sub></td><td align="left" valign="top">0.7887</td><td align="left" valign="top">0.8125</td><td align="left" valign="top">0.7428</td><td align="left" valign="top">0.7761</td></tr><tr><td align="left" valign="top">LR<sub>multimodal fusion</sub></td><td align="left" valign="top">0.8582</td><td align="left" valign="top">0.8558</td><td align="left" valign="top">0.8563</td><td align="left" valign="top">0.8542</td></tr><tr><td align="left" valign="top">RF<sup><xref ref-type="table-fn" rid="table6fn4">d</xref></sup><sub>early fusion</sub></td><td align="left" valign="top">0.8028</td><td align="left" valign="top">0.8889</td><td align="left" valign="top">0.6857</td><td align="left" valign="top">0.7742</td></tr><tr><td align="left" valign="top">RF<sub>late fusion</sub></td><td align="left" valign="top">0.7887</td><td align="left" valign="top">0.8333</td><td align="left" valign="top">0.7142</td><td align="left" valign="top">0.7692</td></tr><tr><td align="left" valign="top">Neural network</td><td align="left" valign="top">0.8311</td><td align="left" valign="top">0.8964</td><td align="left" valign="top">0.7427</td><td align="left" valign="top">0.8099</td></tr><tr><td align="left" valign="top">RF<sub>multimodal fusion</sub></td><td align="left" valign="top">0.8437</td><td align="left" valign="top">0.8509</td><td align="left" valign="top">0.8275</td><td align="left" valign="top">0.8371</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>Performance comparison between early fusion (feature concatenation), late fusion (decision-level fusion of text and audio), a neural network baseline trained on concatenated embeddings, and the proposed attention-based multimodal fusion (text + audio + image) on the ADReSSo 2021 unseen test set (n=71) is shown.</p></fn><fn id="table6fn2"><p><sup>b</sup>SVC: support vector classifier.</p></fn><fn id="table6fn3"><p><sup>c</sup>LR: logistic regression.</p></fn><fn id="table6fn4"><p><sup>d</sup>RF: random forest.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-4"><title>Bidirectional Cross-Attention Versus Cross-Attention Only</title><p>We evaluate 2 fusion variants to quantify the impact of the bidirectional cross attention. Across all 3 classifiers (<xref ref-type="table" rid="table7">Table 7</xref>), the bidirectional cross-attention configuration achieves stronger overall performance than cross-attention only. For SVC, cross-attention alone attains an <italic>F</italic><sub>1</sub>-score of 0.8571 with accuracy of 0.8732, while our bidirectional cross attention improves results to an <italic>F</italic><sub>1</sub>-score of 0.8667 and an accuracy of 0.8722. A similar trend is observed for LR. For RF, the cross-attention only baseline achieves an <italic>F</italic><sub>1</sub>-score of 0.8125 (accuracy=0.8309), and the bidirectional cross-attention variant improves to an <italic>F</italic><sub>1</sub>-score of 0.8371 (accuracy=0.8437). Overall, bidirectional cross-attention produces balanced AD classification performance metrics.</p><table-wrap id="t7" position="float"><label>Table 7.</label><caption><p>Performance comparison between cross-attention&#x2013;only fusion and the proposed bidirectional cross-attention fusion on the ADReSSo 2021 unseen test set (n=71) across 3 classifiers (support vector classifier [SVC], logistic regression [LR], random forest [RF]).</p></caption><table id="table7" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Fusion method</td><td align="left" valign="bottom">Accuracy</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top">SVC<sub>cross_attention</sub></td><td align="left" valign="top">0.8732</td><td align="left" valign="top">0.9642</td><td align="left" valign="top">0.7714</td><td align="left" valign="top">0.8571</td></tr><tr><td align="left" valign="top">SVC<sub>bidirectional</sub></td><td align="left" valign="top">0.8722</td><td align="left" valign="top">0.8806</td><td align="left" valign="top">0.8564</td><td align="left" valign="top">0.8667</td></tr><tr><td align="left" valign="top">LR<sub>cross_attention</sub></td><td align="left" valign="top">0.8028</td><td align="left" valign="top">0.7837</td><td align="left" valign="top">0.8285</td><td align="left" valign="top">0.8055</td></tr><tr><td align="left" valign="top">LR<sub>bidirectional</sub></td><td align="left" valign="top">0.8582</td><td align="left" valign="top">0.8558</td><td align="left" valign="top">0.8563</td><td align="left" valign="top">0.8542</td></tr><tr><td align="left" valign="top">RF<sub>cross_attention</sub></td><td align="left" valign="top">0.8309</td><td align="left" valign="top">0.8965</td><td align="left" valign="top">0.7428</td><td align="left" valign="top">0.8125</td></tr><tr><td align="left" valign="top">RF<sub>bidirectional</sub></td><td align="left" valign="top">0.8437</td><td align="left" valign="top">0.8509</td><td align="left" valign="top">0.8275</td><td align="left" valign="top">0.8371</td></tr></tbody></table></table-wrap></sec><sec id="s3-5"><title>AD Severity Assessment Task</title><sec id="s3-5-1"><title>Overview</title><p>We evaluated how well individual modalities support continuous MMSE score prediction and compare them to pairwise and their multimodal counterparts (<xref ref-type="table" rid="table8">Table 8</xref>). Across all 3 modalities for regression, text embeddings again outperform audio. With ModernBERT features, Ridge regression attains the lowest text-only error, followed closely by SVR and RFR. In contrast, the corresponding audio-only models with wav2vec2 embeddings are substantially less accurate.</p><table-wrap id="t8" position="float"><label>Table 8.</label><caption><p>Mini-Mental Status Examination (MMSE) score prediction across unimodal, bimodal, and trimodal fusion settings on the ADReSSo 2021 unseen test set (n=71)<sup><xref ref-type="table-fn" rid="table8fn1">a</xref></sup>.</p></caption><table id="table8" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model and classifier</td><td align="left" valign="bottom">RMSE<sup><xref ref-type="table-fn" rid="table8fn2">b</xref></sup> (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Unimodal</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SVR<sup><xref ref-type="table-fn" rid="table8fn3">c</xref></sup><sub>text</sub></td><td align="left" valign="top">4.3959 (3.7870&#x2010;5.0445)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SVR<sub>audio</sub></td><td align="left" valign="top">6.0146 (4.6954&#x2010;7.2511)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ridge<sup><xref ref-type="table-fn" rid="table8fn4">d</xref></sup><sub>text</sub></td><td align="left" valign="top">4.5951 (3.7248&#x2010;5.4198)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ridge<sub>audio</sub></td><td align="left" valign="top">5.9914 (4.9110&#x2010;7.1118)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RFR<sup><xref ref-type="table-fn" rid="table8fn5">e</xref></sup><sub>text</sub></td><td align="left" valign="top">4.5806 (3.8630&#x2010;5.3138)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RFR<sub>audio</sub></td><td align="left" valign="top">8.0297 (6.8131&#x2010;9.2356)</td></tr><tr><td align="left" valign="top" colspan="2">Pairwise</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SVR<sub>text + image</sub></td><td align="left" valign="top">4.3897 (3.7793&#x2010;4.9916)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ridge<sub>text + image</sub></td><td align="left" valign="top">4.3333 (3.5186&#x2010;4.7430)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RFR<sub>text + image</sub></td><td align="left" valign="top">4.4310 (3.6517&#x2010;5.2476)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SVR<sub>audio + image</sub></td><td align="left" valign="top">6.0201 (4.6987&#x2010;7.2746)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ridge<sub>audio + image</sub></td><td align="left" valign="top">5.8611 (4.9170&#x2010;6.8695)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RFR<sub>audio + image</sub></td><td align="left" valign="top">6.1440 (5.0956&#x2010;7.1654)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SVR<sub>text + audio</sub></td><td align="left" valign="top">4.6745 (3.9837&#x2010;5.3725)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ridge<sub>text + audio</sub></td><td align="left" valign="top">4.3122 (3.7134&#x2010;4.9219)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RFR<sub>text + audio</sub></td><td align="left" valign="top">4.7432 (4.0348&#x2010;5.4675)</td></tr><tr><td align="left" valign="top" colspan="2">Multimodal</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>SVR<sub>text + audio + image</sub></td><td align="left" valign="top">4.2134 (3.4898&#x2010;5.0544)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ridge<sub>text + audio + image</sub></td><td align="left" valign="top">4.1440 (3.5511&#x2010;4.7421)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>RFR<sub>text + audio + image</sub></td><td align="left" valign="top">4.3856 (3.8126&#x2010;5.0012)</td></tr></tbody></table><table-wrap-foot><fn id="table8fn1"><p><sup>a</sup>Root mean squared error with 95% CI is reported for unimodal models using text-only (ModernBERT embeddings) and audio-only (Wav2Vec 2.0 embeddings), bimodal fusion models using text + image, text + audio, and audio + image representations, and the proposed trimodal fusion model using text + audio + image. The results are shown for 3 regression heads: support vector regressor, ridge regression, and random forest regressor.</p></fn><fn id="table8fn2"><p><sup>b</sup>RMSE: root mean squared error.</p></fn><fn id="table8fn3"><p><sup>c</sup>SVR: support vector regressor.</p></fn><fn id="table8fn4"><p><sup>d</sup>Ridge: ridge regression.</p></fn><fn id="table8fn5"><p><sup>e</sup>RFR: random forest regressor.</p></fn></table-wrap-foot></table-wrap><p>For pairwise fusion, the lowest prediction error is achieved when text is included in the fusion pair, while combinations that exclude text remain substantially weaker. The best-performing bimodal configuration is Ridge with text + image, which attains the lowest RMSE of 4.3333, followed closely by SVR. These results indicate that integrating a second modality alongside text can reduce prediction error relative to text-only baselines. In contrast, pairings based on audio + image perform consistently worse across all regression heads.</p><p>Across the 3 regression heads, multimodal fusion achieves consistently low error, with Ridge obtaining the best overall RMSE of 4.1440, followed closely by SVR regression. RFR also performs competitively with an RMSE of 4.3856. Overall, the 3-way fusion setting yields strong and stable MMSE prediction accuracy across model families, with SVR and Ridge producing the lowest errors.</p></sec><sec id="s3-5-2"><title>Comparison With Representative ADReSSo 2021 Multimodal and Transformer-Based Studies</title><p>To further position our method relative to recent benchmark-relevant approaches, we conducted a targeted scoping review of ADReSSo 2021 studies using multimodal, transformer, and attention-based methods. From this review, we selected e representative ADReSSo 2021 comparator studies for focused comparison: Wang et al [<xref ref-type="bibr" rid="ref28">28</xref>], Zhu et al [<xref ref-type="bibr" rid="ref29">29</xref>], Bang et al [<xref ref-type="bibr" rid="ref30">30</xref>], and Shao and Fang [<xref ref-type="bibr" rid="ref31">31</xref>]. These studies span attention-based multimodal fusion, wav2vec2 + Bidirectional Encoder Representations from Transformers (BERT) hybrid modeling, large language model&#x2013;augmented multimodal modeling, and co-attention acoustic-text fusion, respectively. <xref ref-type="table" rid="table9">Table 9</xref> summarizes their reported AD classification performance alongside our proposed model. Details of the scoping-review procedure are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><table-wrap id="t9" position="float"><label>Table 9.</label><caption><p>Comparison with representative modern ADReSSo 2021 multimodal and transformer-based comparator models for Alzheimer disease (AD) classification<sup><xref ref-type="table-fn" rid="table9fn1">a</xref></sup>.</p></caption><table id="table9" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study and year</td><td align="left" valign="bottom">Modalities</td><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Accuracy</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top">Wang et al [<xref ref-type="bibr" rid="ref28">28</xref>], 2021</td><td align="left" valign="top">Linguistic + IS10 + X-Vector</td><td align="left" valign="top">DNN<sup><xref ref-type="table-fn" rid="table9fn2">b</xref></sup></td><td align="left" valign="top">0.80</td><td align="left" valign="top">0.83</td></tr><tr><td align="left" valign="top">Zhu et al [<xref ref-type="bibr" rid="ref29">29</xref>], 2021</td><td align="left" valign="top">Audio + ASR<sup><xref ref-type="table-fn" rid="table9fn3">c</xref></sup> text</td><td align="left" valign="top">Wav2vec2 + BERT<sup><xref ref-type="table-fn" rid="table9fn4">d</xref></sup></td><td align="left" valign="top">0.83</td><td align="left" valign="top">0.83</td></tr><tr><td align="left" valign="top">Bang et al [<xref ref-type="bibr" rid="ref30">30</xref>], 2024</td><td align="left" valign="top">Audio + text + LLM<sup><xref ref-type="table-fn" rid="table9fn5">e</xref></sup>-generated &#x201C;opinion&#x201D; feature</td><td align="left" valign="top">ChatGPT-assisted multimodal model</td><td align="left" valign="top">0.87</td><td align="left" valign="top">0.87</td></tr><tr><td align="left" valign="top">Shao and Fang [<xref ref-type="bibr" rid="ref31">31</xref>], 2025</td><td align="left" valign="top">Acoustic + ASR text</td><td align="left" valign="top">Co-attention multimodal model</td><td align="left" valign="top">0.83</td><td align="left" valign="top">0.84</td></tr><tr><td align="left" valign="top">Ours</td><td align="left" valign="top">Audio + text + image</td><td align="left" valign="top">Bidirectional attention + SVC</td><td align="left" valign="top">0.87</td><td align="left" valign="top">0.87</td></tr></tbody></table><table-wrap-foot><fn id="table9fn1"><p><sup>a</sup>The table reports a focused comparison between our proposed audio + text + image bidirectional attention framework and four representative ADReSSo 2021 studies selected from the targeted scoping review: Wang et al [<xref ref-type="bibr" rid="ref28">28</xref>] (attention-based multimodal deep neural network), Zhu et al [<xref ref-type="bibr" rid="ref29">29</xref>] (wav2vec2 + BERT), Bang et al [<xref ref-type="bibr" rid="ref30">30</xref>] (large language model&#x2013;augmented multimodal model), and Shao and Fang [<xref ref-type="bibr" rid="ref31">31</xref>] (co-attention multimodal fusion). Accuracy and <italic>F</italic><sub>1</sub>-score are reported as provided in the respective studies.</p></fn><fn id="table9fn2"><p><sup>b</sup>DNN: deep neural network.</p></fn><fn id="table9fn3"><p><sup>c</sup>ASR: automatic speech recognition.</p></fn><fn id="table9fn4"><p><sup>d</sup>BERT: Bidirectional Encoder Representations from Transformers.</p></fn><fn id="table9fn5"><p><sup>e</sup>LLM: large language model.</p></fn></table-wrap-foot></table-wrap></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><p>Our findings indicate that trimodal fusion of text, audio, and image provides the overall strong performance on ADReSSo for both AD classification and MMSE prediction. Relative to unimodal and bimodal alternatives, the trimodal setting yields the best overall classification performance and the lowest or near-lowest regression error across model families. Because ADReSSo was designed to reduce common confounds such as demographic imbalance, these gains are likely to reflect disease-relevant information rather than spurious correlations [<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>A consistent trend across the tasks is that text representations dominate unimodal performance, with transcript-derived embeddings outperforming audio-only models by a substantial margin. This aligns with longstanding clinical and computational evidence that connected speech contains measurable dementia-related changes in lexical selection, semantic content, and syntactic organization [<xref ref-type="bibr" rid="ref32">32</xref>]. In our experiments, ModernBERT-based embeddings preserve much of this discriminative linguistic structure, yielding the strong unimodal baselines across both classification and MMSE prediction.</p><p>Although audio-only performance is weaker, audio contributes useful complementary information when paired with text, particularly for AD classification where text + audio is a rather competitive bimodal configuration and for MMSE prediction. This is consistent with prior work on paralinguistic and timing-related markers in dementia, such as pausing, hesitation, and fluency disruptions, which are not fully captured in transcripts alone [<xref ref-type="bibr" rid="ref33">33</xref>]. The use of self-supervised acoustic representations such as wav2vec 2.0 provides a principled way to encode these cues from raw speech [<xref ref-type="bibr" rid="ref19">19</xref>].</p><p>The image modality plays a more nuanced role in this benchmark. Because the Cookie Theft stimulus is fixed across participants, an image-only system is not expected to be discriminative; however, our results suggest that including image embeddings can still improve performance when used jointly with text and audio. One plausible explanation is that the image representation acts as a shared contextual anchor for the narrative. This interpretation is supported by several findings. First, drawing on the image-text matching approach [<xref ref-type="bibr" rid="ref34">34</xref>], prior work [<xref ref-type="bibr" rid="ref35">35</xref>] shows that healthy and dementia samples differ in their relevance to the target picture, with healthy samples yielding higher relevance scores. This finding suggests that visual information can contribute meaningfully to dementia detection. Second, we conducted a series of comparisons to evaluate the added value of the image modality (<xref ref-type="table" rid="table1">Tables 1</xref><xref ref-type="table" rid="table2"/><xref ref-type="table" rid="table3"/><xref ref-type="table" rid="table4"/>-<xref ref-type="table" rid="table5">5</xref>), including text versus text + image, audio versus audio + image, and text + audio versus text + audio + image. Across all settings, incorporating the image leads to consistent, albeit modest, performance gains when combined with other modalities. Third, we performed a control analysis by randomly shuffling half of the image patches in the text + image fusion model. This manipulation resulted in a drop in performance relative to the original images (<italic>F</italic><sub>1</sub>-score: 0.84 vs 0.79), providing direct evidence that preserving the visual structure is important for model effectiveness. Taken together, these results indicate that the image serves as a semantic anchor to leverage the full contextual information available. Since the same image is used for every participant, the study demonstrates the usefulness of shared stimulus grounding rather than the predictive value of participant-specific image data.</p><p>A key methodological advance from our ablation study is that bidirectional cross-attention consistently outperforms standard 1-way cross-attention across classifier families. This suggests that reciprocal information exchange between modalities is more effective than a single one-way update for this task. In other words, allowing each modality to attend to and be updated by the other leads to stronger fusion than unidirectional conditioning alone [<xref ref-type="bibr" rid="ref12">12</xref>]. The comparison with early and late fusion further shows that how modalities are integrated matters. Early fusion and late fusion are simple and widely used, but they do not model cross-modal relationships as explicitly as attention-based intermediate fusion. Their weaker performance in our experiments supports the value of learned cross-modal interaction over heuristic fusion strategies [<xref ref-type="bibr" rid="ref35">35</xref>].</p><p>Across tasks, the regression results mirror classification trends: text-only is the strongest among unimodal baselines, bimodal improvements are most pronounced when text is included, and trimodal fusion produces the lowest overall error. This consistency suggests that the multimodal embedding learned through attention mechanisms supports both discrete diagnostic labeling and continuous cognitive assessment, a useful property for clinical decision support where disease manifestations often exist on a spectrum rather than a binary boundary. At the same time, RMSE values remain nontrivial, highlighting that cognitive score prediction remains a challenging objective likely influenced by heterogeneity in impairment profiles and the limited size of available labeled data.</p><p>Beyond the in-benchmark comparisons reported in the <italic>Results</italic> section, <xref ref-type="table" rid="table9">Table 9</xref> provides additional context against representative modern ADReSSo 2021 multimodal and transformer-based studies. As shown in <xref ref-type="table" rid="table9">Table 9</xref>, our method remains competitive relative to these comparator models. At the same time, direct one-to-one comparison remains imperfect even within the ADReSSo-only subset because studies differ in modality definitions, auxiliary cues, and evaluation protocols. We therefore use <xref ref-type="table" rid="table9">Table 9</xref> as a focused benchmark-context comparison rather than as a strict ranked state-of-the-art table.</p><p>Several limitations should be considered when interpreting these findings. First, the findings have not yet been validated across an independent dataset, a different elicitation image, or a different clinical population. ADReSSo is a relatively small benchmark dataset for training and evaluating multimodal deep learning models, with 237 total samples and a held-out test set of 71 recordings. Although the dataset is carefully curated and demographically balanced, the limited sample size may constrain generalizability and increase the possibility that strong performance partly reflects benchmark-specific characteristics rather than broader robustness across populations or recording conditions. Accordingly, this study should be interpreted primarily as a benchmarked methodological investigation on ADReSSo that evaluates the potential of bidirectional multimodal fusion, rather than as definitive evidence of clinical generalization. Second, picture-description tasks represent a structured elicitation paradigm; performance may differ for more naturalistic conversational speech, where topic drift and dialogue dynamics introduce additional complexity. Third, transcript-based modeling depends on automatic speech recognition quality; while Whisper-style transcription is robust in many conditions, transcription choices around disfluencies and fillers can influence downstream embedding behavior especially in tasks where such markers may carry clinical signal.</p><p>Finally, these results suggest several directions for future work. More fine-grained alignment strategies (eg, segment-level coupling between audio and text representations) may strengthen the ability of cross-attention to learn clinically meaningful correspondences. Additionally, evaluation across multiple elicitation tasks and datasets would help clarify whether the observed gains extend beyond Cookie Theft.</p><p>In this work, we introduced an attention-based trimodal fusion framework for dementia screening that integrates text, audio, and image through bidirectional attention. Evaluated on the ADReSSo Cookie Theft picture-description benchmark, the proposed approach achieved overall strong performance for both AD classification and MMSE prediction, outperforming unimodal baselines, bimodal fusion models, and conventional early- and late-fusion strategies. The ablation analysis further showed that bidirectional cross-attention performs better than standard 1-way cross-attention, highlighting the value of reciprocal cross-modal interaction for dementia-related prediction.</p></sec></body><back><ack><p>We are grateful for the ADReSSo Challenge data that were available via DementiaBank. This work was supported by computational resources provided by the Centre for Large AI Models (CLAIM) of The Hong Kong Polytechnic University. ChatGPT was used for basic error correction (grammar, typos, and editing) in the initial draft for the purpose of rephrasing or rewording.</p></ack><notes><sec><title>Funding</title><p>Research reported in this publication was supported by the National Institute on Aging of the National Institutes of Health under Award Number P30AG073105. The content is solely the responsibility of the authors and does not necessarily represent the official views of the National Institutes of Health. Additional support is provided by PolyU Strategic Hiring Scheme and Faculty Reserve.</p></sec><sec><title>Data Availability</title><p>All the data are available online [<xref ref-type="bibr" rid="ref36">36</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: FA, HL</p><p>Formal analysis: FA</p><p>Investigation: FA</p><p>Methodology: FA, HL</p><p>Supervision: HL</p><p>Writing &#x2013; original draft: FA, HL</p><p>Writing &#x2013; review and editing: FA, HL</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AD</term><def><p>Alzheimer disease</p></def></def-item><def-item><term id="abb2">AUC</term><def><p>area under the receiver operating characteristic curve</p></def></def-item><def-item><term id="abb3">BERT</term><def><p>Bidirectional Encoder Representations from Transformers</p></def></def-item><def-item><term id="abb4">CLIP</term><def><p>contrastive language&#x2013;image pretraining</p></def></def-item><def-item><term id="abb5">LR</term><def><p>logistic regression</p></def></def-item><def-item><term id="abb6">MMSE</term><def><p>Mini-Mental Status Examination</p></def></def-item><def-item><term id="abb7">RF</term><def><p>random forest</p></def></def-item><def-item><term id="abb8">RFR</term><def><p>random forest regressor</p></def></def-item><def-item><term id="abb9">Ridge</term><def><p>ridge regression</p></def></def-item><def-item><term id="abb10">RMSE</term><def><p>root mean squared error</p></def></def-item><def-item><term id="abb11">ROC</term><def><p>receiver operating characteristic</p></def></def-item><def-item><term id="abb12">SVC</term><def><p>support vector classifier</p></def></def-item><def-item><term id="abb13">SVR</term><def><p>support vector regressor</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McKhann</surname><given-names>GM</given-names> </name><name name-style="western"><surname>Knopman</surname><given-names>DS</given-names> </name><name name-style="western"><surname>Chertkow</surname><given-names>H</given-names> </name><etal/></person-group><article-title>The diagnosis of dementia due to Alzheimer&#x2019;s disease: recommendations from the National Institute on Aging-Alzheimer&#x2019;s Association workgroups on diagnostic guidelines for Alzheimer&#x2019;s disease</article-title><source>Alzheimers Dement</source><year>2011</year><month>05</month><volume>7</volume><issue>3</issue><fpage>263</fpage><lpage>269</lpage><pub-id pub-id-type="doi">10.1016/j.jalz.2011.03.005</pub-id><pub-id pub-id-type="medline">21514250</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Taler</surname><given-names>V</given-names> </name><name name-style="western"><surname>Phillips</surname><given-names>NA</given-names> </name></person-group><article-title>Language performance in Alzheimer&#x2019;s disease and mild cognitive impairment: a comparative review</article-title><source>J Clin Exp Neuropsychol</source><year>2008</year><month>07</month><volume>30</volume><issue>5</issue><fpage>501</fpage><lpage>556</lpage><pub-id pub-id-type="doi">10.1080/13803390701550128</pub-id><pub-id pub-id-type="medline">18569251</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Livingston</surname><given-names>G</given-names> </name><name name-style="western"><surname>Huntley</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sommerlad</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Dementia prevention, intervention, and care: 2020 report of the Lancet Commission</article-title><source>Lancet</source><year>2020</year><month>08</month><day>8</day><volume>396</volume><issue>10248</issue><fpage>413</fpage><lpage>446</lpage><pub-id pub-id-type="doi">10.1016/S0140-6736(20)30367-6</pub-id><pub-id pub-id-type="medline">32738937</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Folstein</surname><given-names>MF</given-names> </name><name name-style="western"><surname>Folstein</surname><given-names>SE</given-names> </name><name name-style="western"><surname>McHugh</surname><given-names>PR</given-names> </name></person-group><article-title>&#x201C;Mini-mental state&#x201D;. A practical method for grading the cognitive state of patients for the clinician</article-title><source>J Psychiatr Res</source><year>1975</year><month>11</month><volume>12</volume><issue>3</issue><fpage>189</fpage><lpage>198</lpage><pub-id pub-id-type="doi">10.1016/0022-3956(75)90026-6</pub-id><pub-id pub-id-type="medline">1202204</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jack</surname><given-names>CR</given-names>  <suffix>Jr</suffix></name><name name-style="western"><surname>Andrews</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Beach</surname><given-names>TG</given-names> </name><etal/></person-group><article-title>Revised criteria for diagnosis and staging of Alzheimer&#x2019;s disease: Alzheimer&#x2019;s Association Workgroup</article-title><source>Alzheimers Dement</source><year>2024</year><month>08</month><volume>20</volume><issue>8</issue><fpage>5143</fpage><lpage>5169</lpage><pub-id pub-id-type="doi">10.1002/alz.13859</pub-id><pub-id pub-id-type="medline">38934362</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Agbavor</surname><given-names>F</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>H</given-names> </name></person-group><article-title>Predicting dementia from spontaneous speech using large language models</article-title><source>PLOS Digit Health</source><year>2022</year><month>12</month><volume>1</volume><issue>12</issue><fpage>e0000168</fpage><pub-id pub-id-type="doi">10.1371/journal.pdig.0000168</pub-id><pub-id pub-id-type="medline">36812634</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Eyigoz</surname><given-names>E</given-names> </name><name name-style="western"><surname>Mathur</surname><given-names>S</given-names> </name><name name-style="western"><surname>Santamaria</surname><given-names>M</given-names> </name><name name-style="western"><surname>Cecchi</surname><given-names>G</given-names> </name><name name-style="western"><surname>Naylor</surname><given-names>M</given-names> </name></person-group><article-title>Linguistic markers predict onset of Alzheimer&#x2019;s disease</article-title><source>EClinicalMedicine</source><year>2020</year><month>11</month><volume>28</volume><fpage>100583</fpage><pub-id pub-id-type="doi">10.1016/j.eclinm.2020.100583</pub-id><pub-id pub-id-type="medline">33294808</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Luz</surname><given-names>S</given-names> </name><name name-style="western"><surname>Haider</surname><given-names>F</given-names> </name><name name-style="western"><surname>Fuente</surname><given-names>S de la</given-names> </name><name name-style="western"><surname>Fromm</surname><given-names>D</given-names> </name><name name-style="western"><surname>MacWhinney</surname><given-names>B</given-names> </name></person-group><article-title>Detecting cognitive decline using speech only: the ADReSSo challenge</article-title><conf-name>INTERSPEECH 2021</conf-name><conf-date>Aug 30 to Sep 3, 2021</conf-date><pub-id pub-id-type="doi">10.21437/Interspeech.2021-1220</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ksibi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Walha</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zakariah</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ayadi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Alshalali</surname><given-names>T</given-names> </name><name name-style="western"><surname>Almujally</surname><given-names>NA</given-names> </name></person-group><article-title>Multimodal Siamese networks for dementia detection from speech in women</article-title><source>Sci Rep</source><year>2025</year><month>08</month><day>22</day><volume>15</volume><issue>1</issue><fpage>30938</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-13902-7</pub-id><pub-id pub-id-type="medline">40847098</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ilias</surname><given-names>L</given-names> </name><name name-style="western"><surname>Askounis</surname><given-names>D</given-names> </name></person-group><article-title>Multimodal deep learning models for detecting dementia from speech and transcripts</article-title><source>Front Aging Neurosci</source><year>2022</year><volume>14</volume><fpage>830943</fpage><pub-id pub-id-type="doi">10.3389/fnagi.2022.830943</pub-id><pub-id pub-id-type="medline">35370608</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Tsai</surname><given-names>YHH</given-names> </name><name name-style="western"><surname>Bai</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yamada</surname><given-names>M</given-names> </name><name name-style="western"><surname>Morency</surname><given-names>LP</given-names> </name><name name-style="western"><surname>Salakhutdinov</surname><given-names>R</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Inui</surname><given-names>K</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>X</given-names> </name></person-group><article-title>Transformer dissection: an unified understanding for transformer&#x2019;s attention via the lens of kernel</article-title><source>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP</source><year>2019</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>4344</fpage><lpage>4353</lpage><pub-id pub-id-type="doi">10.18653/v1/D19-1443</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Atrey</surname><given-names>PK</given-names> </name><name name-style="western"><surname>Hossain</surname><given-names>MA</given-names> </name><name name-style="western"><surname>El Saddik</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kankanhalli</surname><given-names>MS</given-names> </name></person-group><article-title>Multimodal fusion for multimedia analysis: a survey</article-title><source>Multimedia Systems</source><year>2010</year><month>11</month><volume>16</volume><issue>6</issue><fpage>345</fpage><lpage>379</lpage><pub-id pub-id-type="doi">10.1007/s00530-010-0182-0</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Arevalo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Solorio</surname><given-names>T</given-names> </name><name name-style="western"><surname>Montes-y-G&#x00F3;mez</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gonz&#x00E1;lez</surname><given-names>FA</given-names> </name></person-group><article-title>Gated multimodal units for information fusion</article-title><source>arXiv</source><comment>Preprint posted online on  Feb 7, 2017</comment><pub-id pub-id-type="doi">10.48550/arXiv.1702.01992</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Vaswani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shazeer</surname><given-names>N</given-names> </name><name name-style="western"><surname>Parmar</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Attention is all you need</article-title><access-date>2026-09-10</access-date><conf-name>31st Conference on Neural Information Processing Systems (NIPS 2017)</conf-name><conf-date>Dec 4-9, 2017</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2017/file/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf">https://proceedings.neurips.cc/paper_files/paper/2017/file/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Tan</surname><given-names>H</given-names> </name><name name-style="western"><surname>Bansal</surname><given-names>M</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Inui</surname><given-names>K</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>X</given-names> </name></person-group><article-title>LXMERT: learning cross-modality encoder representations from transformers</article-title><source>Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)</source><year>2019</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>5100</fpage><lpage>5111</lpage><pub-id pub-id-type="doi">10.18653/v1/D19-1514</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Agbavor</surname><given-names>F</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>H</given-names> </name></person-group><article-title>Dementia detection from spontaneous speech using cross-attention fusion</article-title><source>J Dement Alzheimers Dis</source><year>2026</year><volume>3</volume><issue>1</issue><fpage>12</fpage><pub-id pub-id-type="doi">10.3390/jdad3010012</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Goodglass</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kaplan</surname><given-names>E</given-names> </name><name name-style="western"><surname>Barresi</surname><given-names>B</given-names> </name></person-group><source>The Assessment of Aphasia and Related Disorders</source><year>2001</year><edition>3</edition><publisher-name>Lippincott Williams &#x0026; Wilkins</publisher-name><pub-id pub-id-type="other">9780683305593</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Becker</surname><given-names>JT</given-names> </name><name name-style="western"><surname>Boller</surname><given-names>F</given-names> </name><name name-style="western"><surname>Lopez</surname><given-names>OL</given-names> </name><name name-style="western"><surname>Saxton</surname><given-names>J</given-names> </name><name name-style="western"><surname>McGonigle</surname><given-names>KL</given-names> </name></person-group><article-title>The natural history of Alzheimer&#x2019;s disease. Description of study cohort and accuracy of diagnosis</article-title><source>Arch Neurol</source><year>1994</year><month>06</month><volume>51</volume><issue>6</issue><fpage>585</fpage><lpage>594</lpage><pub-id pub-id-type="doi">10.1001/archneur.1994.00540180063015</pub-id><pub-id pub-id-type="medline">8198470</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Baevski</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Mohamed</surname><given-names>A</given-names> </name><name name-style="western"><surname>Auli</surname><given-names>M</given-names> </name></person-group><article-title>Wav2vec 2.0: a framework for self-supervised learning of speech representations</article-title><access-date>2022-07-14</access-date><conf-name>34th Conference on Neural Information Processing Systems (NeurIPS 2020)</conf-name><conf-date>Dec 6-12, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper/2020/hash/92d1e1eb1cd6f9fba3227870bb6d7f07-Abstract.html">https://proceedings.neurips.cc/paper/2020/hash/92d1e1eb1cd6f9fba3227870bb6d7f07-Abstract.html</ext-link></comment></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Radford</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Brockman</surname><given-names>G</given-names> </name><name name-style="western"><surname>McLeavey</surname><given-names>C</given-names> </name><name name-style="western"><surname>Sutskever</surname><given-names>I</given-names> </name></person-group><article-title>Robust speech recognition via large-scale weak supervision</article-title><access-date>2026-09-10</access-date><conf-name>ICML&#x2019;23: Proceedings of the 40th International Conference on Machine Learning</conf-name><conf-date>Jul 23-29, 2023</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/10.5555/3618408.3619590">https://dl.acm.org/doi/10.5555/3618408.3619590</ext-link></comment></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Warner</surname><given-names>B</given-names> </name><name name-style="western"><surname>Chaffin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Clavi&#x00E9;</surname><given-names>B</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Che</surname><given-names>W</given-names> </name><name name-style="western"><surname>Nabende</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shutova</surname><given-names>E</given-names> </name><name name-style="western"><surname>Pilehvar</surname><given-names>MT</given-names> </name></person-group><article-title>Smarter, better, faster, longer: a modern bidirectional encoder for fast, memory efficient, and long context finetuning and inference</article-title><source>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)</source><year>2025</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>2526</fpage><lpage>2547</lpage><pub-id pub-id-type="doi">10.18653/v1/2025.acl-long.127</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alishiri</surname><given-names>GH</given-names> </name><name name-style="western"><surname>Bayat</surname><given-names>N</given-names> </name><name name-style="western"><surname>Fathi Ashtiani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Tavallaii</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Assari</surname><given-names>S</given-names> </name><name name-style="western"><surname>Moharamzad</surname><given-names>Y</given-names> </name></person-group><article-title>Logistic regression models for predicting physical and mental health-related quality of life in rheumatoid arthritis patients</article-title><source>Mod Rheumatol</source><year>2008</year><month>12</month><volume>18</volume><issue>6</issue><fpage>601</fpage><lpage>608</lpage><pub-id pub-id-type="doi">10.3109/s10165-008-0092-6</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Biau</surname><given-names>G</given-names> </name></person-group><article-title>Analysis of a random forests model</article-title><source>J Mach Learn Res</source><year>2012</year><access-date>2026-09-10</access-date><volume>13</volume><fpage>1063</fpage><lpage>1095</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://jmlr.org/papers/v13/biau12a.html">https://jmlr.org/papers/v13/biau12a.html</ext-link></comment></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pedregosa</surname><given-names>F</given-names> </name><name name-style="western"><surname>Varoquaux</surname><given-names>G</given-names> </name><name name-style="western"><surname>Gramfort</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Scikit-learn: machine learning in Python</article-title><source>J Mach Learn Res</source><year>2011</year><access-date>2026-09-10</access-date><volume>12</volume><fpage>2825</fpage><lpage>2830</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.jmlr.org/papers/volume12/pedregosa11a/pedregosa11a.pdf">https://www.jmlr.org/papers/volume12/pedregosa11a/pedregosa11a.pdf</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Sch&#x00F6;lkopf</surname><given-names>B</given-names> </name><name name-style="western"><surname>Smola</surname><given-names>AJ</given-names> </name></person-group><source>Learning with Kernels: Support Vector Machines, Regularization, Optimization, and Beyond</source><year>2001</year><publisher-name>The MIT Press</publisher-name><pub-id pub-id-type="doi">10.7551/mitpress/4175.001.0001</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hoerl</surname><given-names>AE</given-names> </name><name name-style="western"><surname>Kennard</surname><given-names>RW</given-names> </name></person-group><article-title>Ridge regression: biased estimation for nonorthogonal problems</article-title><source>Technometrics</source><year>1970</year><month>02</month><volume>12</volume><issue>1</issue><fpage>55</fpage><lpage>67</lpage><pub-id pub-id-type="doi">10.1080/00401706.1970.10488634</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hinton</surname><given-names>GE</given-names> </name><name name-style="western"><surname>Salakhutdinov</surname><given-names>RR</given-names> </name></person-group><article-title>Reducing the dimensionality of data with neural networks</article-title><source>Science</source><year>2006</year><month>07</month><day>28</day><volume>313</volume><issue>5786</issue><fpage>504</fpage><lpage>507</lpage><pub-id pub-id-type="doi">10.1126/science.1127647</pub-id><pub-id pub-id-type="medline">16873662</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>N</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hao</surname><given-names>S</given-names> </name><name name-style="western"><surname>Shao</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Subbalakshmi</surname><given-names>KP</given-names> </name></person-group><article-title>Modular multi-modal attention network for Alzheimer&#x2019;s disease detection using patient audio and language data</article-title><conf-name>Interspeech 2021</conf-name><conf-date>Aug 30 to Sep 3, 2021</conf-date><pub-id pub-id-type="doi">10.21437/Interspeech.2021-2024</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Obyat</surname><given-names>A</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Batsis</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Roth</surname><given-names>RM</given-names> </name></person-group><article-title>WavBERT: exploiting semantic and non-semantic speech using wav2vec and BERT for dementia detection</article-title><conf-name>Interspeech 2021</conf-name><conf-date>Aug 30 to Sep 3, 2021</conf-date><pub-id pub-id-type="doi">10.21437/Interspeech.2021-332</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bang</surname><given-names>JU</given-names> </name><name name-style="western"><surname>Han</surname><given-names>SH</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>BO</given-names> </name></person-group><article-title>Alzheimer&#x2019;s disease recognition from spontaneous speech using large language models</article-title><source>ETRI Journal</source><year>2024</year><month>02</month><volume>46</volume><issue>1</issue><fpage>96</fpage><lpage>105</lpage><pub-id pub-id-type="doi">10.4218/etrij.2023-0356</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Shao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Fang</surname><given-names>T</given-names> </name></person-group><article-title>Alzheimer&#x2019;s disease detection using co-attention mechanism for acoustic and ASR-transcribed text features</article-title><conf-name>Interspeech 2025</conf-name><conf-date>Aug 17-21, 2025</conf-date><pub-id pub-id-type="doi">10.21437/Interspeech.2025-219</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ahmed</surname><given-names>S</given-names> </name><name name-style="western"><surname>Haigh</surname><given-names>AMF</given-names> </name><name name-style="western"><surname>de Jager</surname><given-names>CA</given-names> </name><name name-style="western"><surname>Garrard</surname><given-names>P</given-names> </name></person-group><article-title>Connected speech as a marker of disease progression in autopsy-proven Alzheimer&#x2019;s disease</article-title><source>Brain</source><year>2013</year><month>12</month><volume>136</volume><issue>Pt 12</issue><fpage>3727</fpage><lpage>3737</lpage><pub-id pub-id-type="doi">10.1093/brain/awt269</pub-id><pub-id pub-id-type="medline">24142144</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>H</given-names> </name><name name-style="western"><surname>Karjadi</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ang</surname><given-names>TFA</given-names> </name><etal/></person-group><article-title>Identification of digital voice biomarkers for cognitive health</article-title><source>Explor Med</source><year>2020</year><volume>1</volume><fpage>406</fpage><lpage>417</lpage><pub-id pub-id-type="doi">10.37349/emed.2020.00028</pub-id><pub-id pub-id-type="medline">33665648</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Radford</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Hallacy</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Learning transferable visual models from natural language supervision</article-title><access-date>2026-09-10</access-date><conf-name>Proceedings of the 38th International Conference on Machine Learning</conf-name><conf-date>Jul 18-24, 2021</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v139/radford21a/radford21a.pdf">https://proceedings.mlr.press/v139/radford21a/radford21a.pdf</ext-link></comment></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>N</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Batsis</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Roth</surname><given-names>RM</given-names> </name><name name-style="western"><surname>MacWhinney</surname><given-names>B</given-names> </name></person-group><article-title>Evaluating picture description speech for dementia detection using image-text alignment</article-title><source>ACM Trans Comput Healthcare</source><year>2025</year><pub-id pub-id-type="doi">10.1145/3777482</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="web"><article-title>DementiaBank</article-title><source>TalkBank</source><access-date>2026-09-10</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://talkbank.org/dementia/">https://talkbank.org/dementia/</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Details of the scoping review procedure.</p><media xlink:href="aging_v9i1e93279_app1.docx" xlink:title="DOCX File, 23 KB"/></supplementary-material></app-group></back></article>