<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.0 20120330//EN" "JATS-journalpublishing1.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">INFORMATICA</journal-id>
<journal-title-group><journal-title>Informatica</journal-title></journal-title-group>
<issn pub-type="epub">1822-8844</issn><issn pub-type="ppub">0868-4952</issn><issn-l>0868-4952</issn-l>
<publisher>
<publisher-name>Vilnius University</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">INFOR643</article-id>
<article-id pub-id-type="doi">10.15388/26-INFOR643</article-id>
<article-categories><subj-group subj-group-type="heading">
<subject>Research Article</subject></subj-group></article-categories>
<title-group>
<article-title>Frequency Domain Decoupling and Semantic Filtering Network for Multimodal Sentiment Analysis</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Liu</surname><given-names>Yundong</given-names></name><email xlink:href="szxylyd@126.com">szxylyd@126.com</email><xref ref-type="aff" rid="j_infor643_aff_001">1</xref><bio>
<p><bold>Y. Liu</bold> received the master of science degree in educational technology from East China Normal University, China. He is currently an associate professor in the Computer Science and Technology program at Suzhou University, China. His research interests include image recognition and artificial intelligence.</p></bio>
</contrib>
<contrib contrib-type="author">
<name><surname>Tan</surname><given-names>Chengfang</given-names></name><email xlink:href="874036730@qq.com">874036730@qq.com</email><xref ref-type="aff" rid="j_infor643_aff_001">1</xref><bio>
<p><bold>C. Tan</bold> received the master of science degree in educational technology from Nanjing Normal University, China. She is currently an associate professor in the Computer Science and Technology program at Suzhou University, China. Her research interests include image recognition and artificial intelligence.</p></bio>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname><given-names>Shunxiang</given-names></name><email xlink:href="sxzhang@aust.edu.cn">sxzhang@aust.edu.cn</email><xref ref-type="aff" rid="j_infor643_aff_002">2</xref><bio>
<p><bold>S. Zhang</bold> received the PhD degree from the School of Computer Engineering and Science, Shanghai University, Shanghai, China, in 2012. He is currently a professor with the School of Computer Science and Engineering, Anhui University of Science and Technology, Huainan, China. His research interests include intelligent information processing, data mining, big data analytics, and sentiment analysis.</p></bio>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname><given-names>Yulei</given-names></name><email xlink:href="2230817302@qq.com">2230817302@qq.com</email><xref ref-type="aff" rid="j_infor643_aff_002">2</xref><bio>
<p><bold>Y. Zhang</bold> received the master of science degree from Anhui University of Science and Technology, Huainan, China. He is currently with the School of Computer Science and Engineering, Anhui University of Science and Technology, Huainan, China. His research interests include multimodal sentiment analysis, multimodal sarcasm detection, deep learning, and multimodal information fusion.</p></bio>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname><given-names>Kuan-Ching</given-names></name><email xlink:href="edge4xt@outlook.com">edge4xt@outlook.com</email><xref ref-type="aff" rid="j_infor643_aff_003">3</xref><xref ref-type="corresp" rid="cor1">∗</xref><bio>
<p><bold>K.-C. Li</bold> received the PhD degree in electrical engineering from the University of São Paulo, Brazil. He is currently a distinguished professor with the Department of Computer Science and Information Engineering, Providence University, Taiwan. He is also affiliated with the School of Mathematics and Big Data, Anhui University of Science and Technology, Huainan, China. His research interests include cloud computing, GPU computing, big data, and parallel programming.</p></bio>
</contrib>
<contrib contrib-type="author">
<name><surname>Crespo</surname><given-names>Rubén González</given-names></name><email xlink:href="ruben.gonzalez@unir.net">ruben.gonzalez@unir.net</email><xref ref-type="aff" rid="j_infor643_aff_004">4</xref><xref ref-type="corresp" rid="cor1">∗</xref><bio>
<p><bold>R. Crespo</bold> received the PhD degree in computer science engineering from Universidad Pontificia de Salamanca, Spain. He is currently a vice-rector and full professor of Computer Science and Artificial Intelligence at Universidad Internacional de La Rioja (UNIR), Spain. His research interests include artificial intelligence, Industry 4.0, project management, and accessibility.</p></bio>
</contrib>
<aff id="j_infor643_aff_001"><label>1</label><institution>School of Artificial Intelligence and Big Data, Suzhou University</institution>, Suzhou, <country>China</country></aff>
<aff id="j_infor643_aff_002"><label>2</label><institution>School of Computer Science and Engineering, Anhui University of Science and Technology</institution>, Huainan, <country>China</country></aff>
<aff id="j_infor643_aff_003"><label>3</label><institution>School of Mathematics and Big Data, Anhui University of Science and Technology</institution>, Huainan, <country>China</country></aff>
<aff id="j_infor643_aff_004"><label>4</label><institution>Department of Computer Science and Technology, Universidad Internacional de La Rioja</institution>, Logroño, La Rioja, <country>Spain</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>∗</label>Corresponding authors.</corresp>
</author-notes>
<pub-date pub-type="ppub"><year>2026</year></pub-date><pub-date pub-type="epub"><day>18</day><month>8</month><year>2026</year></pub-date><volume content-type="ahead-of-print">0</volume><issue>0</issue><fpage>1</fpage><lpage>29</lpage><history><date date-type="received"><month>6</month><year>2025</year></date><date date-type="accepted"><month>8</month><year>2026</year></date></history>
<permissions><copyright-statement>© 2026 Vilnius University</copyright-statement><copyright-year>2026</copyright-year>
<license license-type="open-access" xlink:href="http://creativecommons.org/licenses/by/4.0/">
<license-p>Open access article under the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">CC BY</ext-link> license.</license-p></license></permissions>
<abstract>
<p>Multimodal sentiment analysis, due to its comprehensive ability to capture user sentiment, has significant application value in areas such as public opinion analysis. Existing research, however, falls short in several aspects: (1) it inadequately models the global structural information of the image, and (2) it overlooks the potential noise impact within each modality. These limitations hinder the accurate extraction of sentiment cues from individual modalities. To address these issues, we propose a Frequency Domain Decoupling and Semantic Filtering Network for Multimodal Sentiment Analysis. This network primarily integrates frequency-domain decoupling with semantic filtering to process high- and low-frequency image information separately, thereby enhancing model performance. Specifically, we designed a Dynamic Frequency Domain Decoupling Module that applies discrete wavelet transforms for differentiated image processing. This module, combined with a Dual-Domain Loss Function, constrains consistency between the text semantic space and the frequency distribution of the optimized image features, preventing sentiment information loss from excessive filtering. The module also incorporates two key components: a Global Semantic Sentiment Component (GSSC) and a High-Frequency Filtering Component (HFFC). In the GSSC component, we designed a Hybrid Mamba to leverage text in capturing global semantic information from low-frequency image data. Furthermore, our HFFC component generates a dynamic weight matrix guided by text, enabling quantitative noise suppression. Additionally, we developed a Multi-Grained Semantic Purification Module to filter noise at the word, phrase, and sentence levels. Experimental findings from publicly accessible datasets indicate that our proposed model achieves competitive performance compared with existing methods on multimodal sentiment analysis under the adopted experimental settings and sarcasm detection tasks, validating the effectiveness of our noise-suppression method in cross-modal sentiment analysis. A key limitation of the model is its use of DWT: downsampling-related resolution loss in low-frequency subbands and independent subband partitioning compromise the capture of large-scale global structural correlations and local-global feature modelling, which advanced transform techniques can alleviate.</p>
</abstract>
<kwd-group>
<label>Key words</label>
<kwd>sentiment analysis</kwd>
<kwd>multimodal</kwd>
<kwd>frequency domain</kwd>
<kwd>semantic filtering</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="j_infor643_s_001">
<label>1</label>
<title>Introduction</title>
<p>Sentiment is a subjective response to external stimuli, and sentiment analysis primarily aims to understand human sentiment using existing knowledge (Pandey and Vishwakarma, <xref ref-type="bibr" rid="j_infor643_ref_043">2024</xref>; Singh <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_048">2024</xref>). With the widespread use of social media, there has been an increasing number of ways to combine image and text to express sentiment. As a result, multimodal sentiment analysis has gained significant attention from researchers (Sun <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_051">2025</xref>; Zhu <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_088">2023</xref>). Compared to traditional text-based sentiment analysis, multimodal sentiment analysis requires handling more modalities, posing greater challenges to researchers (Gandhi <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_017">2023</xref>). Multimodal sentiment analysis has broad potential in social-media analysis, education, healthcare, and other human-centered applications (Gandhi <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_017">2023</xref>; Do <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_013">2024</xref>; Lu <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_037">2024</xref>; Das and Singh, <xref ref-type="bibr" rid="j_infor643_ref_010">2023</xref>). By jointly modelling textual and visual evidence, these methods can capture sentiment cues that may be incomplete or ambiguous in either modality alone (Zhu <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_090">2022</xref>). More broadly, data-driven intelligent technologies have also been applied to shipping communication security, privacy-preserving collaborative learning, and secure maritime data management (Feng <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_015">2026</xref>; Li <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_031">2025b</xref>,a).</p>
<p>In the past few years, considerable advances have been made in multimodal sentiment analysis (Wang <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_059">2025a</xref>; Zhao <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_081">2025a</xref>; Wang <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_063">2025c</xref>). Early research primarily focused on performing sentiment analysis by examining the correlation between images and texts, which played a crucial role in improving sentiment classification accuracy (Zhao <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_083">2019</xref>; Truong and Lauw, <xref ref-type="bibr" rid="j_infor643_ref_055">2019</xref>). Other studies have introduced cross-modal mechanisms, such as multi-level attention mechanisms and information correlation modelling, to model images globally (Xue <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_072">2022</xref>; Chen <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_006">2023</xref>; Wang <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_060">2025b</xref>), further enhancing the interaction between modalities, or explored the deep fusion of multimodal features through Transformer technologies (Kim and Park, <xref ref-type="bibr" rid="j_infor643_ref_027">2023</xref>; Aziz <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_002">2025</xref>; Wu <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_067">2025</xref>). Some studies have also incorporated colour cues and cross-modal translation mechanisms for sentiment analysis (An and Zainon, <xref ref-type="bibr" rid="j_infor643_ref_001">2023</xref>; Zhang <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_076">2024</xref>). Despite significant advancements in inter-modal interaction modelling, Kim <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_028">2021</xref>) argue that networks dealing with images should be more complex than those for text. However, existing research has not deeply explored the impact of the global structure of images on sentiment understanding, nor has it adequately addressed the interference of latent noise within modalities during sentiment feature extraction. This noise may originate from irrelevant background information in the image or from redundant expressions in the text, hindering the precise extraction of sentiment features and reducing the model’s accuracy (Zhou <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_087">2024</xref>, <xref ref-type="bibr" rid="j_infor643_ref_085">2026a</xref>, <xref ref-type="bibr" rid="j_infor643_ref_086">2026b</xref>).</p>
<p>To tackle this problem, this work proposes a new approach called <bold>F</bold>requency <bold>D</bold>omain Decoupling and <bold>S</bold>emantic <bold>F</bold>iltering <bold>Net</bold>work for Multimodal Sentiment Analysis (FDSF-Net). The objective is to suppress intra-modal noise interference and enhance the expressive ability of sentiment features within each modality. Specifically, we design a dynamic frequency-domain decoupling module that applies discrete wavelet transforms to the image, dividing its frequency-domain information into Low-Frequency (LF) global semantic sentiment bands and High-Frequency (HF) noise-sensitive bands, allowing us to distinguish between sentiment information and image noise. In this module, we design different components for low and high frequencies to enable differential processing. Specifically, we have developed a global semantic modelling component and an HF filtering component. The global semantic modelling component, leveraging Mamba’s powerful temporal modelling capabilities, captures the image’s structural information in the LF domain, enabling a thorough understanding of its potential sentiment. The HF filtering component primarily filters the HF noise-sensitive domain, thereby improving the model’s ability to capture sentiment cues. To further optimize the frequency-domain decoupling effect, we introduce a dual-domain loss function. This loss function enforces consistency between the text semantic space and frequency-decoupled image features in the frequency domain during optimization, thereby preventing the loss of sentiment information that may result from excessive filtering. This design allows our model to not only accurately extract sentiment features from each modality when processing cross-modal information but also to minimize the influence of noise. Additionally, we have designed a multi-granularity semantic purification module to filter noise at multiple levels in the text. This module filters text information temporally at the word, phrase, and sentence levels, alleviating redundant or irrelevant sentiment information and improving the accuracy of text-based sentiment feature extraction. Finally, we fuse the features of both modalities using the Cross-Spectral Interaction Module. This module, through a cross-frequency-domain interaction mechanism, fully considers the interrelationship between image and text, thereby optimizing the fusion of sentiment features. Experimental validation on publicly available datasets demonstrates the effectiveness of our method.</p>
<p>The contributions of our work are as follows: 
<list>
<list-item id="j_infor643_li_001">
<label>•</label>
<p>To propose a multimodal sentiment analysis model based on frequency domain decoupling and semantic filtering, which improves the model’s sentiment analysis capability by modelling the global structure of the image and analysing noise.</p>
</list-item>
<list-item id="j_infor643_li_002">
<label>•</label>
<p>To design a dynamic frequency domain decoupling module and a multi-granularity semantic purification module. The former processes different frequency subbands of the image differentially, while the latter performs noise suppression at multiple granularities of the text, thereby effectively enhancing the model’s ability to extract sentiment cues.</p>
</list-item>
<list-item id="j_infor643_li_003">
<label>•</label>
<p>The proposed model achieves competitive performance on publicly available datasets under the reported single-run experimental setting.</p>
</list-item>
</list>
</p>
</sec>
<sec id="j_infor643_s_002">
<label>2</label>
<title>Related Work</title>
<sec id="j_infor643_s_003">
<label>2.1</label>
<title>Text Sentiment Analysis</title>
<p>Traditional approaches primarily rely on conventional machine learning algorithms. For example, Moraes <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_040">2013</xref>) conducted a comparison using SVM and ANN for document-level sentiment classification tasks. They used the bag-of-words model for feature selection and weighting, and discussed the classification accuracy of both methods across different contexts within a standard evaluation setting. Shunxiang <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_047">2023</xref>) proposed a model based on sentiment intensity and PU learning. They divided reviews into different subsets based on sentiment intensity and used SCAR and Spy techniques to extract initial positive and negative samples. A semi-supervised PU learning detector was then constructed to iteratively detect fake reviews from incoming streaming data. Some studies focus on improving dictionaries and statistical methods to better understand text for sentiment analysis. For instance, Kang <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_026">2012</xref>) addressed the issue of inadequate sentiment lexicons in restaurant reviews by proposing a new sentiment lexicon (senti-lexicon) and improving the Naive Bayes algorithm. They combined unigrams and bigrams as features, narrowing the accuracy gap between positive and negative sentiment classification, and outperforming SVM and the original Naive Bayes in recall and precision. Wang <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_064">2023b</xref>) proposed an automatic method for generating fine-grained sentiment lexicons by constructing a seed lexicon through sentiment-sentiment transfer, extending the lexicon using graph propagation, and performing multi-information fusion based on neural networks. The resulting FGSL demonstrated strong performance across a range of sentiment analysis tasks. Pashchenko <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_045">2022</xref>), using the NRC sentiment lexicon and unsupervised learning, explored the relationship between sentiment and star ratings in hotel and tourism reviews, finding that customer feedback on different sentiment aspects varied with ratings. However, some studies suggest that relying on a single method may limit the model’s expressive power; therefore, a combination of methods may be more effective (Rodríguez-Ibáñez <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_046">2023</xref>). For example, Bibi <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_004">2022</xref>) presented a framework for unsupervised learning that utilizes concepts and hierarchical clustering to analyse sentiment on Twitter. The results showed that unsupervised learning was comparable to supervised learning techniques. Wang <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_062">2019</xref>) explored Twitter sentiment analysis by combining textual information with sentiment diffusion patterns. They proposed the SentiDiff iterative algorithm, which considered the interaction between both elements to enhance sentiment analysis performance, and used sentiment diffusion patterns for the first time to improve Twitter sentiment analysis. While these methods have made significant contributions to sentiment analysis tasks, the powerful expressive capabilities of deep learning have led many studies to adopt deep learning to address issues in this field (Tai <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_052">2015</xref>; Tang <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_053">2015</xref>; Dai <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_009">2021</xref>). For example, Chen (<xref ref-type="bibr" rid="j_infor643_ref_007">2015</xref>) proposed two convolutional neural network models, Parallel CNN and Deep CNN, based on word2vec word embeddings, to identify question-answer relationship candidates in QA systems. They used convolutional layers to extract semantic features, and pooling and fully connected layers to summarize them. Usama <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_056">2020</xref>) introduced a model based on RNN and CNN with an attention mechanism. They first used CNN to extract sentence features, then applied an attention mechanism to compute contextual weights for these features, and finally input the features and weights into an RNN for sentiment analysis. Tian <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_054">2020</xref>) proposed the SKEP model, which incorporates sentiment masking and three sentiment prediction targets to embed sentiment information into pre-trained representations. The model significantly outperformed baseline models, achieving new best results on most test sets. The widespread use of attention mechanisms has provided new perspectives for sentiment analysis of text. Several studies have combined attention mechanisms to understand sentiment in text. For example, Zhai <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_075">2020</xref>) proposed the Multi-AFM model, which generates contextual representations using gating units, this model was applied to sentiment analysis of educational big data, improving classification performance. Parveen <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_044">2023</xref>) proposed the GARN framework, which integrates RNN and attention mechanisms to extract sentiment features, perform feature selection, and conduct sentiment classification on Twitter. More recently, Zhang <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_078">2025b</xref>) proposed a textual graph representation with syntactic weighting that combines word-position graph structure, syntactic weights, attention, and external knowledge to model implicit sentiment. Beyond sentiment analysis, Zhao <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_082">2025b</xref>) demonstrated that hyperbolic graph attention can integrate hierarchical semantic representations with long-context information for technical keyphrase extraction, illustrating the broader value of graph-based long-context modelling for text understanding.</p>
<p>Despite the significant achievements of deep learning in text sentiment analysis, the widespread use of social media means that sentiment expression is often not limited to text alone. Therefore, extending the powerful capabilities of deep learning to multimodal sentiment analysis by integrating information from different modalities to more accurately identify and understand sentiment is the primary focus of this study.</p>
</sec>
<sec id="j_infor643_s_004">
<label>2.2</label>
<title>Multimodal Sentiment Analysis</title>
<p>Multimodal sentiment analysis poses a greater challenge than traditional text-based sentiment analysis, as it requires integrating information from multiple modalities to uncover underlying sentiment cues. Traditional methods predominantly rely on attention mechanisms to enable models to focus on key information across different modalities, thereby enhancing sentiment analysis performance (Truong and Lauw, <xref ref-type="bibr" rid="j_infor643_ref_055">2019</xref>; Xue <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_072">2022</xref>). For instance, Zhang <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_079">2023</xref>) utilized the Transformer model to extract both image and text features, incorporating LSTM and attention to better emphasize important information. Li <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_032">2023</xref>), Liang <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_034">2025</xref>) introduced tensors to extract and fuse features from different sources, such as text, images, and audio, in order to capture the complex data patterns in multimodal sentiment analysis. Additionally, an improved bilinear fusion method was employed, in which the model dynamically adjusts the fusion weights in real time based on the features of both images and text to achieve weighted fusion. Hu <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_023">2024</xref>) proposed a three-channel multimodal fusion framework where the second channel removes redundant information from auxiliary modalities, while the third channel enhances the significance of the primary modality, and the integration of features from the three channels is managed by a multi-channel information fusion gate. Wang <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_060">2025b</xref>) introduced an adaptive attention module that dynamically adjusts the contributions of text and image features by leveraging cross-modal attention to extract shared representations, thereby improving the performance of both modalities. Additionally, sentiment information is used to guide the extraction of features relevant to sentiment. Several studies have focused on effectively fusing features from different modalities to leverage their complementary information (Zhao <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_083">2019</xref>; Wang <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_057">2022</xref>). For example, Mai <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_038">2022</xref>) employed intra-modal, cross-modal, and semi-contrastive learning concurrently, thoroughly exploring cross-modal interactions and learning the relationships between samples and categories, thereby reducing the modality gap. Moreover, they introduced refinement terms and modality intervals to learn single-modal pairs better. An and Zainon (<xref ref-type="bibr" rid="j_infor643_ref_001">2023</xref>) enhanced sentiment analysis accuracy by integrating semantic information and image colour cues from image-text pairs. The model included a feature extraction module to extract semantic and colour features, a feature interaction module to facilitate information exchange between features via cross-attention mechanisms, and a label prediction module to consolidate features and strengthen multimodal sentiment analysis. Cheng <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_008">2023</xref>) proposed an Attention Temporal Convolution Network (ATCN) to enhance the representation of temporal features in a single modality, alongside a Multi-layer Feature Fusion (MFF) model to improve multimodal fusion performance, fused features of different levels based on feature correlation, and used cross-modal multi-head attention to explore relationships among low-level features.</p>
<p>In addition, semantic correlations and sentiment consistency between different modalities are crucial for sentiment analysis, leading some studies to incorporate contrastive learning to enhance model performance. For instance, Zhang <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_076">2024</xref>) used a translation network encoder to capture shared concepts between vision and text, addressing the issue of missing modalities. Based on this framework, a single-modal weight adaptation strategy was introduced to leverage meta-learning from a few labelled samples to learn single-modal weights. Wang <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_059">2025a</xref>) explores modality correlations by a multi-layer cross-modal interaction module. Subsequently, feature fusion was performed using a multimodal fusion module that incorporated contrastive learning to uncover key sentiment features across different categories and samples.</p>
<p>With the growing use of cutting-edge methods, including generative and graph models, more research is focusing on these areas. For example, Huan <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_024">2023</xref>) proposed a UniMF model, which includes a translation module and a prediction module. The translation module generates missing modalities using existing modal information through Multimodal Generation Masking (MGM) and a Multimodal Generation Transformer (MGT). The prediction module integrates multimodal information via an attention mechanism to generate predictions, using a Multimodal Understanding Transformer (MUT) that incorporates Multimodal Understanding Masks (MUM) and Multimodal Sequence (MMSeq) representations for unified multimodal understanding. Zhao <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_081">2025a</xref>) generated virtual modalities to replace missing ones and aligned the semantic space of virtual and missing modalities through contrastive loss. Wang <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_058">2023a</xref>) addressed sentiment information in multimodal data from both global and local fine-grained perspectives. The global perspective obtained overall sentiment representations from text-image caption pairs, while the local perspective explored fine-grained information in text and images via two graph structures. Ultimately, combining global and local sentiment information provided aspect-level sentiment polarity. Wang <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_065">2025d</xref>) captures sentiment correlations within and across modalities through a graph-based architecture. More recently, Zhang <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_077">2025a</xref>) proposed a Multimodal Semantic Fusion Network that uses gated attention for cross-modal alignment, graph convolutional networks to model interactions among aligned features, and the integration of explicit and implicit sentiment semantics.</p>
<p>Although these methods have made progress in multimodal sentiment analysis tasks, they have not fully exploited the global structural information in images and have overlooked noise interference during cross-modal interactions. Given that multimodal data often contains substantial redundancy and noise, these interferences can disrupt the model’s training and degrade performance. Thus, the effective differentiation of image processing and noise mitigation between image and text data to improve the accuracy of multimodal sentiment analysis is the primary focus of this study.</p>
</sec>
</sec>
<sec id="j_infor643_s_005">
<label>3</label>
<title>Proposed Method</title>
<p>This section presents the architecture and core components of the proposed FDSF-Net framework in Fig. <xref rid="j_infor643_fig_001">1</xref>.</p>
<fig id="j_infor643_fig_001">
<label>Fig. 1</label>
<caption>
<p>Framework of FDSF-Net. The model first employs ViT and RoBERTa to extract visual features <inline-formula id="j_infor643_ineq_001"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${Y_{b,c,l}}$]]></tex-math></alternatives></inline-formula> and textual features <inline-formula id="j_infor643_ineq_002"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${T_{b,s,d}}$]]></tex-math></alternatives></inline-formula>, where <italic>b</italic> denotes the batch index, <italic>c</italic> the number of image patches, <italic>l</italic> the visual embedding dimension, <italic>s</italic> the text sequence length, and <italic>d</italic> the textual embedding dimension. The Dynamic Frequency Domain Decoupling Module (dashed box 1) decomposes input images into low-frequency (LF) and high-frequency (HF) components via discrete wavelet transform (DWT), and reconstructs optimized visual features <inline-formula id="j_infor643_ineq_003"><alternatives><mml:math>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">final</mml:mtext>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[${\mathbf{Y}_{b,c,l}^{\textit{final}}}$]]></tex-math></alternatives></inline-formula> using inverse DWT (IDWT). The Multi-Grained Semantic Purification Module (dashed box 2) refines <italic>T</italic> into denoised textual features <inline-formula id="j_infor643_ineq_004"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">s</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${T_{s}}$]]></tex-math></alternatives></inline-formula> by suppressing noise at word-, phrase-, and sentence-level granularities. The Cross-Spectral Interaction Module (dashed box 3) performs bidirectional cross-modal attention between <inline-formula id="j_infor643_ineq_005"><alternatives><mml:math>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">final</mml:mtext>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[${\mathbf{Y}_{b,c,l}^{\textit{final}}}$]]></tex-math></alternatives></inline-formula> and <inline-formula id="j_infor643_ineq_006"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">s</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${T_{s}}$]]></tex-math></alternatives></inline-formula>, producing the fused representation <inline-formula id="j_infor643_ineq_007"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">fused</mml:mtext>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${F_{\textit{fused}}}$]]></tex-math></alternatives></inline-formula> for sentiment prediction <inline-formula id="j_infor643_ineq_008"><alternatives><mml:math><mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">ˆ</mml:mo></mml:mover></mml:math><tex-math><![CDATA[$\hat{y}$]]></tex-math></alternatives></inline-formula>. Solid arrows indicate the direction of information flow, while dashed boxes highlight the three core modules.</p>
</caption>
<graphic xlink:href="infor643_g001.jpg"/>
</fig>
<sec id="j_infor643_s_006">
<label>3.1</label>
<title>Task Definition and Description</title>
<p>Multimodal sentiment analysis primarily aims to detect the sentiment conveyed across different modalities. The multimodal data in this study consists of image and text. Given a text-image pair <inline-formula id="j_infor643_ineq_009"><alternatives><mml:math>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[$({X_{t}^{j}},{X_{i}^{j}})$]]></tex-math></alternatives></inline-formula> along with the corresponding task label <inline-formula id="j_infor643_ineq_010"><alternatives><mml:math>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${y^{j}}$]]></tex-math></alternatives></inline-formula>, the dataset is defined as follows: 
<disp-formula id="j_infor643_eq_001">
<label>(1)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mi mathvariant="italic">D</mml:mi>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mo fence="true" maxsize="1.19em" minsize="1.19em">{</mml:mo>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo fence="true" maxsize="1.19em" minsize="1.19em">}</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">N</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ D={\big\{\big({X_{t}^{j}},{X_{i}^{j}}\big),{y^{j}}\big\}_{j=1}^{N}},\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_011"><alternatives><mml:math>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[${X_{t}^{j}}$]]></tex-math></alternatives></inline-formula> represents the <italic>j</italic>-th text sample in the dataset, <inline-formula id="j_infor643_ineq_012"><alternatives><mml:math>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[${X_{i}^{j}}$]]></tex-math></alternatives></inline-formula> is the corresponding image associated with <inline-formula id="j_infor643_ineq_013"><alternatives><mml:math>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[${X_{t}^{j}}$]]></tex-math></alternatives></inline-formula>, and <inline-formula id="j_infor643_ineq_014"><alternatives><mml:math>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${y^{j}}$]]></tex-math></alternatives></inline-formula> denotes the task-specific label. Specifically, for our core task of multimodal sentiment analysis, <inline-formula id="j_infor643_ineq_015"><alternatives><mml:math>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${y^{j}}$]]></tex-math></alternatives></inline-formula> is a binary label for the HFM dataset, with 1 indicating positive sentiment and 0 indicating negative sentiment, whereas it is a 3-dimensional one-hot vector for the MVSA datasets, indicating negative, neutral, and positive sentiment, respectively. To further improve the generalization ability of our proposed method, we also conduct experiments on multimodal sarcasm-detection datasets, where the labels are split into sarcasm and non-sarcasm. The mapping between these sarcasm detection labels and sentiment labels follows the approach of Wei <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_066">2023</xref>): sarcasm labels are mapped to negative sentiment, and non-sarcasm labels are mapped to positive sentiment. <italic>N</italic> refers to the total number of samples in the dataset, and the text-image pair <inline-formula id="j_infor643_ineq_016"><alternatives><mml:math>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[$({X_{t}^{j}},{X_{i}^{j}})$]]></tex-math></alternatives></inline-formula> is fed into the model <italic>F</italic> to predict the task-specific label, which realizes the corresponding multimodal classification goal: 
<disp-formula id="j_infor643_eq_002">
<label>(2)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo stretchy="false">⟼</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ F\big({X_{t}^{j}},{X_{i}^{j}}\big)\longmapsto {y^{j}}.\]]]></tex-math></alternatives>
</disp-formula> 
This formulation describes the prediction process of the model for each text-image pair in the dataset.</p>
</sec>
<sec id="j_infor643_s_007">
<label>3.2</label>
<title>Feature Extraction Module</title>
<sec id="j_infor643_s_008">
<label>3.2.1</label>
<title>Text Feature Extraction</title>
<p>A pre-trained model, RoBERTa-base (Liu, <xref ref-type="bibr" rid="j_infor643_ref_036">2019</xref>), trained on large-scale datasets, is used as the text encoder. First, the text is split into a sequence of tokens, and then these sequences are input into RoBERTa to obtain text features <inline-formula id="j_infor643_ineq_017"><alternatives><mml:math>
<mml:mi mathvariant="bold">T</mml:mi>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">B</mml:mi>
<mml:mo>×</mml:mo>
<mml:mi mathvariant="italic">S</mml:mi>
<mml:mo>×</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[$\mathbf{T}\in {\mathbb{R}^{B\times S\times {d_{t}}}}$]]></tex-math></alternatives></inline-formula>, where <italic>B</italic> is the batch size, <italic>S</italic> is the token-sequence length, and <inline-formula id="j_infor643_ineq_018"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>768</mml:mn></mml:math><tex-math><![CDATA[${d_{t}}=768$]]></tex-math></alternatives></inline-formula>, as represented by the following formula: 
<disp-formula id="j_infor643_eq_003">
<label>(3)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mi mathvariant="bold">T</mml:mi>
<mml:mo>=</mml:mo>
<mml:mo movablelimits="false">RoBERTa</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo fence="true" stretchy="false">[</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mo movablelimits="false">…</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo fence="true" stretchy="false">]</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ \mathbf{T}=\operatorname{RoBERTa}({\mathbf{X}_{t}})=[{\mathbf{t}_{1}},{\mathbf{t}_{2}},\dots ,{\mathbf{t}_{S}}],\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_019"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">B</mml:mi>
<mml:mo>×</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${\mathbf{t}_{e}}\in {\mathbb{R}^{B\times {d_{t}}}}$]]></tex-math></alternatives></inline-formula> represents the batch of embeddings at token position <italic>e</italic>. The equivalent element-wise notation used below is <inline-formula id="j_infor643_ineq_020"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\mathbf{T}_{b,s,d}}$]]></tex-math></alternatives></inline-formula>.</p>
</sec>
<sec id="j_infor643_s_009">
<label>3.2.2</label>
<title>Image Feature Extraction</title>
<p>We employ the Vision Transformer (ViT) (Dosovitskiy <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_014">2020</xref>) as our image encoder. Initially, the input image <inline-formula id="j_infor643_ineq_021"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${X_{i}}$]]></tex-math></alternatives></inline-formula> with a fixed size of <inline-formula id="j_infor643_ineq_022"><alternatives><mml:math>
<mml:mn>224</mml:mn>
<mml:mo>×</mml:mo>
<mml:mn>224</mml:mn>
<mml:mo>×</mml:mo>
<mml:mn>3</mml:mn></mml:math><tex-math><![CDATA[$224\times 224\times 3$]]></tex-math></alternatives></inline-formula> (<inline-formula id="j_infor643_ineq_023"><alternatives><mml:math>
<mml:mtext mathvariant="italic">height</mml:mtext>
<mml:mo>×</mml:mo>
<mml:mtext mathvariant="italic">width</mml:mtext>
<mml:mo>×</mml:mo>
<mml:mtext mathvariant="italic">channels</mml:mtext></mml:math><tex-math><![CDATA[$\textit{height}\times \textit{width}\times \textit{channels}$]]></tex-math></alternatives></inline-formula>) is segmented into a collection of m-flattened 2D patches. These patches are subsequently fed into ViT to extract image features <inline-formula id="j_infor643_ineq_024"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">m</mml:mi>
<mml:mo>×</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${Y_{v}}\in {\mathbb{R}^{m\times {d_{v}}}}$]]></tex-math></alternatives></inline-formula>, as illustrated in equation (<xref rid="j_infor643_eq_004">4</xref>): 
<disp-formula id="j_infor643_eq_004">
<label>(4)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="italic">V</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
<mml:mi mathvariant="italic">T</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo fence="true" stretchy="false">[</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mo>…</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo fence="true" stretchy="false">]</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {Y_{v}}=ViT({X_{i}})=[{e_{1}},{e_{2}},\dots ,{e_{m}}],\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_025"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${e_{p}}\in {\mathbb{R}^{{d_{v}}}}$]]></tex-math></alternatives></inline-formula> is the embedding of the <italic>p</italic>-th image patch.</p>
</sec>
</sec>
<sec id="j_infor643_s_010">
<label>3.3</label>
<title>Dynamic Frequency Domain Decoupling Module</title>
<fig id="j_infor643_fig_002">
<label>Fig. 2</label>
<caption>
<p>Illustration of the discrete wavelet transform decomposition process used in the proposed frequency-domain decoupling module.. The input image is first subjected to low-pass and high-pass filtering in the horizontal direction, then filtered again in the vertical direction, resulting in four sub-bands: LL, LH, HL, and HH.</p>
</caption>
<graphic xlink:href="infor643_g002.jpg"/>
</fig>
<p>To accurately capture sentiment-related features within an image, our method employs the Discrete Wavelet Transform (DWT) to decompose the image data into specialized frequency domains, thereby preserving both global structure and fine-grained textures (Zhao <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_080">2021</xref>). Specifically, we utilize time-frequency analysis techniques—including wavelets (Mallat, <xref ref-type="bibr" rid="j_infor643_ref_039">1989</xref>; Daubechies and Heil, <xref ref-type="bibr" rid="j_infor643_ref_011">1992</xref>)—to minimize noise while isolating distinct visual components. Through a series of low-pass and high-pass filtering operations, our model decomposes the input into four sub-bands: <inline-formula id="j_infor643_ineq_026"><alternatives><mml:math>
<mml:mi mathvariant="italic">L</mml:mi>
<mml:mi mathvariant="italic">L</mml:mi></mml:math><tex-math><![CDATA[$LL$]]></tex-math></alternatives></inline-formula> (Low-Low), <inline-formula id="j_infor643_ineq_027"><alternatives><mml:math>
<mml:mi mathvariant="italic">L</mml:mi>
<mml:mi mathvariant="italic">H</mml:mi></mml:math><tex-math><![CDATA[$LH$]]></tex-math></alternatives></inline-formula> (Low-High), <inline-formula id="j_infor643_ineq_028"><alternatives><mml:math>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mi mathvariant="italic">L</mml:mi></mml:math><tex-math><![CDATA[$HL$]]></tex-math></alternatives></inline-formula> (High-Low), and <inline-formula id="j_infor643_ineq_029"><alternatives><mml:math>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mi mathvariant="italic">H</mml:mi></mml:math><tex-math><![CDATA[$HH$]]></tex-math></alternatives></inline-formula> (High-High). In our technical framework, the <inline-formula id="j_infor643_ineq_030"><alternatives><mml:math>
<mml:mi mathvariant="italic">L</mml:mi>
<mml:mi mathvariant="italic">L</mml:mi></mml:math><tex-math><![CDATA[$LL$]]></tex-math></alternatives></inline-formula> sub-band constitutes the Low-Frequency (LF) domain, which retains the overall contour and large-scale global information necessary for understanding the primary sentiment tendency (Huang <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_025">2021</xref>). Simultaneously, the combination of <inline-formula id="j_infor643_ineq_031"><alternatives><mml:math>
<mml:mi mathvariant="italic">L</mml:mi>
<mml:mi mathvariant="italic">H</mml:mi></mml:math><tex-math><![CDATA[$LH$]]></tex-math></alternatives></inline-formula>, <inline-formula id="j_infor643_ineq_032"><alternatives><mml:math>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mi mathvariant="italic">L</mml:mi></mml:math><tex-math><![CDATA[$HL$]]></tex-math></alternatives></inline-formula>, and <inline-formula id="j_infor643_ineq_033"><alternatives><mml:math>
<mml:mi mathvariant="italic">H</mml:mi>
<mml:mi mathvariant="italic">H</mml:mi></mml:math><tex-math><![CDATA[$HH$]]></tex-math></alternatives></inline-formula> sub-bands forms the High-Frequency (HF) domain, which focuses on extracting horizontal, vertical, and diagonal edge details. By decoupling the image into these domains, the model can effectively capture subtle visual cues—such as textures and corners—that are often overlooked in spatial analysis but are vital for precise sentiment judgment. The formal decomposition process, illustrated in Fig. <xref rid="j_infor643_fig_002">2</xref>, is expressed as follows: 
<disp-formula id="j_infor643_eq_005">
<label>(5)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="script">DWT</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold">Y</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">ς</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">h</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathbf{Y}_{b,c,l}}=\mathcal{DWT}(\mathbf{Y})=\big({\mathbf{Y}_{b,{c^{\prime }},l}^{\varsigma }},{\mathbf{Y}_{b,{c^{\prime }},l}^{h}}\big),\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_034"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>∗</mml:mo>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${\mathbf{Y}_{b,{c^{\prime }},l}}\in {\mathbb{R}^{\ast }}$]]></tex-math></alternatives></inline-formula> is a three-dimensional tensor, where <italic>b</italic> is the batch index, <italic>c</italic> is the number of patches in the image, <inline-formula id="j_infor643_ineq_035"><alternatives><mml:math>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${c^{\prime }}$]]></tex-math></alternatives></inline-formula> denotes the adjusted number of image patches after decomposition, and <italic>l</italic> is the embedding hidden layer dimension. The wavelet transform decomposes the input image into low-frequency feature <inline-formula id="j_infor643_ineq_036"><alternatives><mml:math>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">ς</mml:mi>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[${\mathbf{Y}_{b,{c^{\prime }},l}^{\varsigma }}$]]></tex-math></alternatives></inline-formula> and high-frequency feature <inline-formula id="j_infor643_ineq_037"><alternatives><mml:math>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">h</mml:mi>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[${\mathbf{Y}_{b,{c^{\prime }},l}^{h}}$]]></tex-math></alternatives></inline-formula>.</p>
<p>In this module, we implement a Global Semantic Sentiment Component (GSSC) and a High-Frequency Filtering Component (HFFC). After processing the GSSC and HFFC components separately, we use the inverse wavelet transform to synthesize the processed LF and HF parts back into the original feature space. The processed LF component, denoted as <inline-formula id="j_infor643_ineq_038"><alternatives><mml:math>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">ς</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="normal">ℑ</mml:mi>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[${\mathbf{Y}_{b,{c^{\prime }},l}^{\varsigma ,\mathrm{\Im }}}$]]></tex-math></alternatives></inline-formula> with the superscript ℑ indicating the output of GSSC, and the processed HF component, denoted as <inline-formula id="j_infor643_ineq_039"><alternatives><mml:math>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">h</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi>ℓ</mml:mi>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[${\mathbf{Y}_{b,{c^{\prime }},l}^{h,\ell }}$]]></tex-math></alternatives></inline-formula> with the superscript <italic>ℓ</italic> indicating the output of HFFC, together reconstruct the final image. This process can be expressed by the following formula: 
<disp-formula id="j_infor643_eq_006">
<label>(6)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">final</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="script">IDWT</mml:mi>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">ς</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="normal">ℑ</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">h</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi>ℓ</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathbf{Y}_{b,c,l}^{\textit{final}}}=\mathcal{IDWT}\big({\mathbf{Y}_{b,{c^{\prime }},l}^{\varsigma ,\mathrm{\Im }}},{\mathbf{Y}_{b,{c^{\prime }},l}^{h,\ell }}\big)\]]]></tex-math></alternatives>
</disp-formula> 
which defines the corresponding representation.</p>
<sec id="j_infor643_s_011">
<label>3.3.1</label>
<title>Global Semantic Sentiment Component</title>
<p>The LF components of the visual modality often carry richer global sentiment semantics. By leveraging ViT’s image sequence modelling capability, we can apply sequence enhancement methods to images. In recent years, state space models have developed rapidly in the field of sequence modelling (Gu <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_021">2021</xref>, <xref ref-type="bibr" rid="j_infor643_ref_020">2022</xref>; Fu <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_016">2022</xref>), attracting the attention of many researchers due to their linear computational complexity and ability to model long-range dependencies. SSMs originated from control theory and model the dynamic evolution process of sequences through state equations, which can be expressed by the following formula: 
<disp-formula id="j_infor643_eq_007">
<label>(7)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="italic">t</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="italic">A</mml:mi>
<mml:mi mathvariant="italic">h</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="italic">t</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="italic">B</mml:mi>
<mml:mi mathvariant="italic">x</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="italic">t</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">y</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="italic">t</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="italic">C</mml:mi>
<mml:mi mathvariant="italic">h</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="italic">t</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ ,{h^{\prime }}(t)=Ah(t)+Bx(t),y(t)=Ch(t),\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_040"><alternatives><mml:math>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>×</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[$\mathbf{A}\in {\mathbb{R}^{{d_{s}}\times {d_{s}}}}$]]></tex-math></alternatives></inline-formula>, <bold>B</bold> and <bold>C</bold> are projection matrices, and <inline-formula id="j_infor643_ineq_041"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">s</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${d_{s}}$]]></tex-math></alternatives></inline-formula> denotes the state dimension, which is different from the dataset size <italic>N</italic> defined in Section <xref rid="j_infor643_s_006">3.1</xref>.</p>
<p>However, Mamba Gu and Dao (<xref ref-type="bibr" rid="j_infor643_ref_019">2023</xref>) further discretizes the parameters <italic>A</italic> and <italic>B</italic> into <inline-formula id="j_infor643_ineq_042"><alternatives><mml:math><mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="italic">A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:math><tex-math><![CDATA[$\bar{A}$]]></tex-math></alternatives></inline-formula> and <inline-formula id="j_infor643_ineq_043"><alternatives><mml:math><mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="italic">B</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:math><tex-math><![CDATA[$\bar{B}$]]></tex-math></alternatives></inline-formula> through the time scale parameter Δ. The discretization calculation can be expressed as: 
<disp-formula id="j_infor643_eq_008">
<label>(8)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="italic">A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">¯</mml:mo></mml:mover>
<mml:mo>=</mml:mo>
<mml:mo movablelimits="false">exp</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="normal">Δ</mml:mi>
<mml:mi mathvariant="italic">A</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo><mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="italic">B</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">¯</mml:mo></mml:mover>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="normal">Δ</mml:mi>
<mml:mi mathvariant="italic">A</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>−</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:mo movablelimits="false">exp</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="normal">Δ</mml:mi>
<mml:mi mathvariant="italic">A</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo>−</mml:mo>
<mml:mi mathvariant="italic">I</mml:mi>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo>·</mml:mo>
<mml:mi mathvariant="normal">Δ</mml:mi>
<mml:mi mathvariant="italic">B</mml:mi>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ \bar{A}=\exp (\Delta A),\bar{B}={(\Delta A)^{-1}}\big(\exp (\Delta A)-I\big)\cdot \Delta B.\]]]></tex-math></alternatives>
</disp-formula> 
At this time, the continuous-time model in (<xref rid="j_infor643_eq_007">7</xref>) yields the following discrete recurrence: 
<disp-formula id="j_infor643_eq_009">
<label>(9)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo><mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">¯</mml:mo></mml:mover>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
<mml:mo>−</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo><mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">B</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">¯</mml:mo></mml:mover>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mspace width="2em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold">C</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathbf{h}_{t}}=\bar{\mathbf{A}}{\mathbf{h}_{t-1}}+\bar{\mathbf{B}}{\mathbf{x}_{t}},\hspace{2em}{\mathbf{y}_{t}}=\mathbf{C}{\mathbf{h}_{t}}.\]]]></tex-math></alternatives>
</disp-formula>
</p>
<p>Mamba has now achieved significant success in deep learning tasks (Ge <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_018">2024</xref>; Pan <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_042">2025</xref>; Xie <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_068">2024</xref>). To effectively enhance the latent sentiment features of an image, we introduce the bidirectional mamba (Zhu <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_089">2024</xref>). This bidirectional mamba not only processes the original features but also performs a flip operation on them, enabling a deeper comprehension of contextual information and uncovering sentiment connections between image and text.</p>
<p>In the meantime, inspired by Ye <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_074">2025</xref>), we employ a collaborative Mamba architecture to handle information interaction between image and text. However, unlike the implementation in Ye <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_074">2025</xref>), we do not use the same A matrix for both image and text. Instead, we introduce a more refined parameterization of Mamba’s core state matrix A, incorporating both a shared component and a modality-specific incremental component. Specifically, we define a cross-modal shared core matrix <inline-formula id="j_infor643_ineq_044"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">shared</mml:mtext>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${A_{\textit{shared}}}$]]></tex-math></alternatives></inline-formula> for the model, which is used to learn the fundamental dynamics and long-term dependencies common across all sequential data. Building upon this, we introduce a sentiment modality-specific incremental matrix <inline-formula id="j_infor643_ineq_045"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">ζ</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${A_{\zeta }}$]]></tex-math></alternatives></inline-formula> and <inline-formula id="j_infor643_ineq_046"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">φ</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${A_{\varphi }}$]]></tex-math></alternatives></inline-formula> for each modality. This can be expressed by the following formula: 
<disp-formula id="j_infor643_eq_010">
<label>(10)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mfenced separators="" open="{" close="">
<mml:mrow>
<mml:mtable equalrows="false" columnlines="none" equalcolumns="false" columnalign="left">
<mml:mtr>
<mml:mtd class="array">
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mstyle mathvariant="bold"><mml:mover accent="true">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">ζ</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mstyle mathvariant="bold"><mml:mover accent="true">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">ζ</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo movablelimits="false">Discretize</mml:mo>
<mml:mspace width="-0.1667em"/>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">Δ</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">ζ</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">shared</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">ζ</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">B</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">ζ</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd class="array">
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mstyle mathvariant="bold"><mml:mover accent="true">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">φ</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mstyle mathvariant="bold"><mml:mover accent="true">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">φ</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo movablelimits="false">Discretize</mml:mo>
<mml:mspace width="-0.1667em"/>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">Δ</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">φ</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">shared</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">φ</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">B</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">φ</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ \left\{\begin{array}{l}({\mathbf{\bar{A}}^{\prime }_{\zeta }},{\mathbf{\bar{B}}^{\prime }_{\zeta }})=\operatorname{Discretize}\hspace{-0.1667em}({\Delta _{\zeta }},{\mathbf{A}_{\mathit{shared}}}+{\mathbf{A}_{\zeta }},{\mathbf{B}_{\zeta }}),\hspace{1em}\\ {} ({\mathbf{\bar{A}}^{\prime }_{\varphi }},{\mathbf{\bar{B}}^{\prime }_{\varphi }})=\operatorname{Discretize}\hspace{-0.1667em}({\Delta _{\varphi }},{\mathbf{A}_{\mathit{shared}}}+{\mathbf{A}_{\varphi }},{\mathbf{B}_{\varphi }}),\hspace{1em}\end{array}\right.\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_047"><alternatives><mml:math>
<mml:mo movablelimits="false">Discretize</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="normal">Δ</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="bold">B</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[$\operatorname{Discretize}(\Delta ,\mathbf{A},\mathbf{B})$]]></tex-math></alternatives></inline-formula> maps the continuous-time pair <inline-formula id="j_infor643_ineq_048"><alternatives><mml:math>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="bold">B</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[$(\mathbf{A},\mathbf{B})$]]></tex-math></alternatives></inline-formula> to <inline-formula id="j_infor643_ineq_049"><alternatives><mml:math>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo><mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">¯</mml:mo></mml:mover>
<mml:mo mathvariant="normal">,</mml:mo><mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">B</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">¯</mml:mo></mml:mover>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[$(\bar{\mathbf{A}},\bar{\mathbf{B}})$]]></tex-math></alternatives></inline-formula> using the input-dependent step size Δ. Thus, both state-space arguments are explicit. After the collaborative Mamba, the image and text representations are concatenated and linearly transformed.</p>
</sec>
<sec id="j_infor643_s_012">
<label>3.3.2</label>
<title>High-Frequency Filtering Component</title>
<p>The HF information in the image often contains noise components, and direct processing may disrupt the continuous semantic information captured by ViT. While ViT captures HF information from an image, it also maintains global semantic consistency. Therefore, directly using the image’s HF information for processing may lead to excessive filtering of these HF components, thereby destroying important sentiment cues. To avoid this issue, we introduce a text-based semantic filtering mechanism that leverages textual semantics to generate a weight matrix, enabling refined suppression of HF noise in images via a frequency-band-aware approach. Additionally, based on the image features filtered by text semantics, we incorporate a residual connection to preserve the image’s core semantic information during filtering, thereby achieving more accurate sentiment analysis. The weighted values of the image HF <inline-formula id="j_infor643_ineq_050"><alternatives><mml:math>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">h</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[${\mathbf{Y}_{b,{c^{\prime }},l}^{h,j}}$]]></tex-math></alternatives></inline-formula> and the text features are calculated through the Multi-Head Attention mechanism. For the Multi-Head Attention mechanism, we can expand it into a weighted sum of multiple heads: 
<disp-formula id="j_infor643_eq_011">
<label>(11)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">hf</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mo largeop="true" movablelimits="false">∑</mml:mo></mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">q</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">H</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">q</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>·</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo movablelimits="false">Attn</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">q</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">hf</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathbf{A}_{b,{c^{\prime }},l}^{\mathrm{hf},j}}={\sum \limits_{q=1}^{H}}{\alpha _{q}}\cdot {\operatorname{Attn}_{q}}\big({\mathbf{Y}_{b,{c^{\prime }},l}^{\mathrm{hf},j}},{\mathbf{T}_{b,s,d}},{\mathbf{T}_{b,s,d}}\big),\]]]></tex-math></alternatives>
</disp-formula> 
where <italic>H</italic> is the number of attention heads, <italic>q</italic> is the summation index, <inline-formula id="j_infor643_ineq_051"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">q</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\alpha _{q}}$]]></tex-math></alternatives></inline-formula> is the weight of head <italic>q</italic>, and the left-hand side does not retain the dummy head index. We obtain the filter matrix <inline-formula id="j_infor643_ineq_052"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">B</mml:mi>
<mml:mo>×</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>×</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${\mathbf{g}_{b,1,{c^{\prime }}}}\in {\mathbb{R}^{B\times 1\times {c^{\prime }}}}$]]></tex-math></alternatives></inline-formula> through a convolution operation, which can be expressed as: 
<disp-formula id="j_infor643_eq_012">
<label>(12)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="italic">σ</mml:mi>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo movablelimits="false">Conv</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">h</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathbf{g}_{b,1,{c^{\prime }}}^{j}}=\sigma \big({\operatorname{Conv}_{1d}}\big({\mathbf{A}_{b,{c^{\prime }},l}^{h,j}}\big)\big),\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_053"><alternatives><mml:math>
<mml:mi mathvariant="italic">σ</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mo>·</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[$\sigma (\cdot )$]]></tex-math></alternatives></inline-formula> denotes the sigmoid function, and <inline-formula id="j_infor643_ineq_054"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mo movablelimits="false">Conv</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mo>·</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[${\operatorname{Conv}_{1d}}(\cdot )$]]></tex-math></alternatives></inline-formula> denotes a one-dimensional convolution operation. Therefore, the denoising of the HF <inline-formula id="j_infor643_ineq_055"><alternatives><mml:math>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">h</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[${\mathbf{Y}_{b,{c^{\prime }},l}^{h,j}}$]]></tex-math></alternatives></inline-formula> can be expressed by the following formula: 
<disp-formula id="j_infor643_eq_013">
<label>(13)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">h</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="false">→</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">h</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">j</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi>ℓ</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">h</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>⊙</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathbf{Y}_{b,{c^{\prime }},l}^{h,j}}\to {\mathbf{Y}_{b,{c^{\prime }},l}^{h,j,\ell }}={\mathbf{Y}_{b,{c^{\prime }},l}^{h}}\odot {\mathbf{g}_{b,{c^{\prime }},1}^{j}}.\]]]></tex-math></alternatives>
</disp-formula> 
This operation represents the denoising process of the high-frequency features. We store multiple HF features in a list, which can be represented as: 
<disp-formula id="j_infor643_eq_014">
<label>(14)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">h</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">j</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi>ℓ</mml:mi>
</mml:mrow>
</mml:msubsup><mml:mover>
<mml:mo stretchy="true">→</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="italic">list</mml:mtext>
</mml:mrow>
</mml:mover>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">h</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi>ℓ</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathbf{Y}_{b,{c^{\prime }},l}^{h,j,\ell }}\xrightarrow{\textit{list}}{\mathbf{Y}_{b,{c^{\prime }},l}^{h,\ell }}.\]]]></tex-math></alternatives>
</disp-formula> 
This formulation indicates that the denoised high-frequency features are aggregated into a list for subsequent processing.</p>
</sec>
</sec>
<sec id="j_infor643_s_013">
<label>3.4</label>
<title>Multi-Grained Semantic Purification Module</title>
<p>In multimodal tasks, text data often contains rich information, and appropriate processing of text is essential for these tasks (Wang <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_061">2024</xref>; Sun <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_050">2024</xref>). However, text often contains various types of noise that manifest at different granularities, negatively affecting the accuracy of sentiment analysis. To effectively address this issue, we designed a Multi-Grained Semantic Purification Module. This module aims to refine text data by filtering out noise at different granularities, thereby extracting more precise sentiment information. To further enhance temporal feature representation, we first apply a causal convolutional operation to extract local temporal dependencies. 
<disp-formula id="j_infor643_eq_015">
<label>(15)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">e</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="script">K</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo>⊙</mml:mo>
<mml:mi mathvariant="italic">σ</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>·</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathbf{T}_{b,s,d}^{e}}=\mathcal{K}({\mathbf{T}_{b,s,d}})\odot \sigma ({\mathbf{W}_{g}}\cdot {\mathbf{T}_{b,s,d}}+{\mathbf{b}_{g}}),\]]]></tex-math></alternatives>
</disp-formula> 
where <italic>s</italic> is the sequence length. The causal convolution operation <inline-formula id="j_infor643_ineq_056"><alternatives><mml:math>
<mml:mi mathvariant="script">K</mml:mi></mml:math><tex-math><![CDATA[$\mathcal{K}$]]></tex-math></alternatives></inline-formula> enhances the representational capacity of temporal features. Subsequently, we generate multi-level feature representations using a multi-grained feature-extraction method that combines multi-head self-attention, convolution, and mean pooling. The formula for calculating the weight distribution for the input text features <inline-formula id="j_infor643_ineq_057"><alternatives><mml:math>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">e</mml:mi>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[${\mathbf{T}_{b,s,d}^{e}}$]]></tex-math></alternatives></inline-formula> is as follows: 
<disp-formula id="j_infor643_eq_016">
<label>(16)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="italic">σ</mml:mi>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:mi mathvariant="italic">L</mml:mi>
<mml:mi mathvariant="italic">N</mml:mi>
<mml:mo fence="true" maxsize="1.19em" minsize="1.19em">[</mml:mo>
<mml:mi mathvariant="script">A</mml:mi>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">e</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="script">C</mml:mi>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">e</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="script">M</mml:mi>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">e</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo fence="true" maxsize="1.19em" minsize="1.19em">]</mml:mo>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathbf{G}_{b,s,d}}=\sigma \big(LN\big[\mathcal{A}\big({\mathbf{T}_{b,s,d}^{e}}\big),\mathcal{C}\big({\mathbf{T}_{b,s,d}^{e}}\big),\mathcal{M}\big({\mathbf{T}_{b,s,d}^{e}}\big)\big]\big),\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_058"><alternatives><mml:math>
<mml:mi mathvariant="italic">σ</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mo>·</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[$\sigma (\cdot )$]]></tex-math></alternatives></inline-formula> denotes the sigmoid activation function, and <inline-formula id="j_infor643_ineq_059"><alternatives><mml:math>
<mml:mo movablelimits="false">LN</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mo>·</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[$\operatorname{LN}(\cdot )$]]></tex-math></alternatives></inline-formula> denotes layer normalization (Ba <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_003">2016</xref>). <inline-formula id="j_infor643_ineq_060"><alternatives><mml:math>
<mml:mi mathvariant="script">A</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[$\mathcal{A}({\mathbf{T}_{b,s,d}})$]]></tex-math></alternatives></inline-formula> denotes the multi-head self-attention operation, <inline-formula id="j_infor643_ineq_061"><alternatives><mml:math>
<mml:mi mathvariant="script">C</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[$\mathcal{C}({\mathbf{T}_{b,s,d}})$]]></tex-math></alternatives></inline-formula> denotes the convolution operation, and <inline-formula id="j_infor643_ineq_062"><alternatives><mml:math>
<mml:mi mathvariant="script">M</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[$\mathcal{M}({\mathbf{T}_{b,s,d}})$]]></tex-math></alternatives></inline-formula> denotes the mean pooling operation. Then, <inline-formula id="j_infor643_ineq_063"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\mathbf{G}_{b,s,d}}$]]></tex-math></alternatives></inline-formula> is used to weight and adjust the features processed by causal convolution, yielding the dynamically adjusted feature representation as follows: 
<disp-formula id="j_infor643_eq_017">
<label>(17)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">s</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">e</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>⊙</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathbf{T}_{b,s,d}^{s}}={\mathbf{T}_{b,s,d}^{e}}\odot {\mathbf{G}_{b,s,d}}.\]]]></tex-math></alternatives>
</disp-formula> 
This operation produces the dynamically adjusted feature representation after applying the learned gating mechanism.</p>
</sec>
<sec id="j_infor643_s_014">
<label>3.5</label>
<title>Cross-Spectral Interaction Module</title>
<p>After processing, crucial information in both image and text features is preserved. To capture the correlation between image and text more precisely, we employ bidirectional cross-modal attention for interaction. Bidirectional cross-modal interaction is modelled in two directions: <inline-formula id="j_infor643_ineq_064"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">→</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">β</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${E_{\alpha }}\to {E_{\beta }}$]]></tex-math></alternatives></inline-formula>, where <inline-formula id="j_infor643_ineq_065"><alternatives><mml:math>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">β</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo stretchy="false">∈</mml:mo>
<mml:mo fence="true" stretchy="false">{</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">final</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">s</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">s</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">final</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo fence="true" stretchy="false">}</mml:mo></mml:math><tex-math><![CDATA[$({E_{\alpha }},{E_{\beta }})\in \{({\mathbf{Y}_{b,c,l}^{\textit{final}}},{\mathbf{T}_{b,s,d}^{s}}),({\mathbf{T}_{b,s,d}^{s}},{\mathbf{Y}_{b,c,l}^{\textit{final}}})\}$]]></tex-math></alternatives></inline-formula>. This approach improves the complementarity between image and text information, enhancing the model’s understanding of deeper relationships between the two modalities. 
<disp-formula id="j_infor643_eq_018">
<label>(18)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true" columnalign="right left" columnspacing="0pt">
<mml:mtr>
<mml:mtd class="align-odd">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd class="align-even">
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo movablelimits="false">Co</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">atts</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">β</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd class="align-odd"/>
<mml:mtd class="align-even">
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo movablelimits="false">softmax</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">β</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" maxsize="2.03em" minsize="2.03em">(</mml:mo><mml:mstyle displaystyle="true">
<mml:mfrac>
<mml:mrow>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">β</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">β</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mstyle>
<mml:mo mathvariant="normal" fence="true" maxsize="2.03em" minsize="2.03em">)</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">β</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">β</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[\begin{aligned}{}{\mathbf{C}_{\alpha }}& ={\operatorname{Co}_{\textit{atts}}}({\mathbf{E}_{\alpha }},{\mathbf{E}_{\beta }})\\ {} & ={\operatorname{softmax}_{\beta }}\bigg(\frac{({\mathbf{W}_{\alpha }}{\mathbf{E}_{\alpha }}){({\mathbf{W}_{\beta }}{\mathbf{E}_{\beta }})^{T}}}{\sqrt{{d_{k}}}}\bigg)({\mathbf{W}_{\beta }}{\mathbf{E}_{\beta }}),\end{aligned}\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_066"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\mathbf{E}_{\alpha }}$]]></tex-math></alternatives></inline-formula> and <inline-formula id="j_infor643_ineq_067"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">β</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\mathbf{E}_{\beta }}$]]></tex-math></alternatives></inline-formula> denote the feature representations of two modalities (e.g. image and text), <inline-formula id="j_infor643_ineq_068"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\mathbf{W}_{\alpha }}$]]></tex-math></alternatives></inline-formula> and <inline-formula id="j_infor643_ineq_069"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">β</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\mathbf{W}_{\beta }}$]]></tex-math></alternatives></inline-formula> are learnable projection matrices, <inline-formula id="j_infor643_ineq_070"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">k</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${d_{k}}$]]></tex-math></alternatives></inline-formula> denotes the scaling dimension, and <inline-formula id="j_infor643_ineq_071"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mo movablelimits="false">Co</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">atts</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mo>·</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[${\operatorname{Co}_{\textit{atts}}}(\cdot )$]]></tex-math></alternatives></inline-formula> denotes the cross-modal attention operation. Simultaneously, to enhance sequence position awareness, we introduce rotary positional embeddings for both the input query and the key (Su <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_049">2024</xref>). Next, we concatenate the processed image features <inline-formula id="j_infor643_ineq_072"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">i</mml:mi>
<mml:mi mathvariant="italic">m</mml:mi>
<mml:mi mathvariant="italic">g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${\mathbf{G}_{img}}\in {\mathbb{R}^{b,c,l}}$]]></tex-math></alternatives></inline-formula>, text features <inline-formula id="j_infor643_ineq_073"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
<mml:mi mathvariant="italic">x</mml:mi>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${\mathbf{G}_{txt}}\in {\mathbb{R}^{b,s,d}}$]]></tex-math></alternatives></inline-formula>, and cross-attention features <inline-formula id="j_infor643_ineq_074"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${\mathbf{C}_{\alpha }}\in {\mathbb{R}^{b,c,l}}$]]></tex-math></alternatives></inline-formula> and <inline-formula id="j_infor643_ineq_075"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">β</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${\mathbf{C}_{\beta }}\in {\mathbb{R}^{b,s,d}}$]]></tex-math></alternatives></inline-formula>. The concatenation operation joins these features column-wise, yielding a new combined feature: 
<disp-formula id="j_infor643_eq_019">
<label>(19)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">γ</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="script">C</mml:mi>
<mml:mo fence="true" stretchy="false">[</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">i</mml:mi>
<mml:mi mathvariant="italic">m</mml:mi>
<mml:mi mathvariant="italic">g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">t</mml:mi>
<mml:mi mathvariant="italic">x</mml:mi>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">β</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo fence="true" stretchy="false">]</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathbf{F}_{\gamma }}=\mathcal{C}[{\mathbf{G}_{img}},{\mathbf{G}_{txt}},{\mathbf{C}_{\alpha }},{\mathbf{C}_{\beta }}],\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_076"><alternatives><mml:math>
<mml:mi mathvariant="script">C</mml:mi></mml:math><tex-math><![CDATA[$\mathcal{C}$]]></tex-math></alternatives></inline-formula> denotes the concatenation operation, and <inline-formula id="j_infor643_ineq_077"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">γ</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${\mathbf{F}_{\gamma }}\in {\mathbb{R}^{b,{s^{\prime }},d}}$]]></tex-math></alternatives></inline-formula> represents the concatenated feature. We then compute the mean of the merged features, fusing the multimodal features into a more compact representation: 
<disp-formula id="j_infor643_eq_020">
<label>(20)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">fused</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo><mml:mstyle displaystyle="true">
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mstyle>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mo largeop="true" movablelimits="false">∑</mml:mo></mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:munderover>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">γ</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mo>:</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">j</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mo>:</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathbf{F}_{\textit{fused}}}=\frac{1}{{S^{\prime }}}{\sum \limits_{j=1}^{{S^{\prime }}}}{\mathbf{F}_{\gamma ,:,j,:}},\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_078"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">fused</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">B</mml:mi>
<mml:mo>×</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${\mathbf{F}_{\textit{fused}}}\in {\mathbb{R}^{B\times {d_{f}}}}$]]></tex-math></alternatives></inline-formula> is the final fused representation, <inline-formula id="j_infor643_ineq_079"><alternatives><mml:math>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="italic">S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>′</mml:mo>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${S^{\prime }}$]]></tex-math></alternatives></inline-formula> is the total number of concatenated positions, and <inline-formula id="j_infor643_ineq_080"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">f</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${d_{f}}$]]></tex-math></alternatives></inline-formula> is the common projected feature dimension. This removes the previously undefined normalizing constant <inline-formula id="j_infor643_ineq_081"><alternatives><mml:math>
<mml:mn>4</mml:mn>
<mml:mi mathvariant="italic">D</mml:mi></mml:math><tex-math><![CDATA[$4D$]]></tex-math></alternatives></inline-formula>.</p>
</sec>
<sec id="j_infor643_s_015">
<label>3.6</label>
<title>Optimization and Classification</title>
<sec id="j_infor643_s_016">
<label>3.6.1</label>
<title>Optimization</title>
<p>For the loss function, we use Focal Loss (Lin <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_035">2017</xref>) as the primary loss, combined with the proposed dual-domain loss as an auxiliary loss. These are integrated through learnable parameters. The Focal Loss function is defined as follows: 
<disp-formula id="j_infor643_eq_021">
<label>(21)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">focal</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">y</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo>−</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>−</mml:mo>
<mml:mo movablelimits="false">softmax</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">γ</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo movablelimits="false">log</mml:mo>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:mo movablelimits="false">softmax</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathcal{L}_{\textit{focal}}}(\mathbf{x},y)=-{\alpha _{y}}{\big(1-\operatorname{softmax}{(\mathbf{x})_{y}}\big)^{\gamma }}\log \big(\operatorname{softmax}{(\mathbf{x})_{y}}\big),\]]]></tex-math></alternatives>
</disp-formula> 
where <bold>x</bold> represents the network output (logits), <italic>y</italic> denotes the ground-truth class label, <inline-formula id="j_infor643_ineq_082"><alternatives><mml:math>
<mml:mo movablelimits="false">softmax</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[$\operatorname{softmax}{(\mathbf{x})_{y}}$]]></tex-math></alternatives></inline-formula> denotes the predicted probability of class <italic>y</italic>, <inline-formula id="j_infor643_ineq_083"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\alpha _{y}}$]]></tex-math></alternatives></inline-formula> is the class-specific balancing factor, and <italic>γ</italic> is the focusing parameter that emphasizes hard-to-classify samples. The final Focal Loss for each sample is calculated as follows: 
<disp-formula id="j_infor643_eq_022">
<label>(22)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">focal</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>−</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">α</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>−</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">γ</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo movablelimits="false">log</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathcal{L}_{\textit{focal}}}=-{\alpha _{y}}{(1-{P_{y}})^{\gamma }}\log {P_{y}},\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_084"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${P_{y}}$]]></tex-math></alternatives></inline-formula> denotes the predicted probability of the ground-truth class. The dual-domain loss comprises a spatial gradient loss and a frequency-domain loss. The spatial term preserves structural information by comparing horizontal and vertical finite differences of the reconstructed and target representations. The frequency-domain loss measures the transformed reconstruction error. The spatial gradient loss is defined as follows: 
<disp-formula id="j_infor643_eq_023">
<label>(23)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">spatial</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mo largeop="true" movablelimits="false">∑</mml:mo></mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">B</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mo fence="true" maxsize="1.19em" minsize="1.19em" stretchy="true">‖</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>∇</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">final</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>−</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>∇</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">target</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo fence="true" maxsize="1.19em" minsize="1.19em" stretchy="true">‖</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mo fence="true" maxsize="1.19em" minsize="1.19em" stretchy="true">‖</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>∇</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">final</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>−</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>∇</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">target</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo fence="true" maxsize="1.19em" minsize="1.19em" stretchy="true">‖</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathcal{L}_{\textit{spatial}}}={\sum \limits_{b=1}^{B}}\big({\big\| {\nabla _{x}}{\mathbf{X}_{b}^{\textit{final}}}-{\nabla _{x}}{\mathbf{X}_{b}^{\textit{target}}}\big\| _{2}^{2}}+{\big\| {\nabla _{y}}{\mathbf{X}_{b}^{\textit{final}}}-{\nabla _{y}}{\mathbf{X}_{b}^{\textit{target}}}\big\| _{2}^{2}}\big),\]]]></tex-math></alternatives>
</disp-formula> 
where <italic>B</italic> denotes the batch size, <inline-formula id="j_infor643_ineq_085"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mo>∇</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">x</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\nabla _{x}}$]]></tex-math></alternatives></inline-formula> and <inline-formula id="j_infor643_ineq_086"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mo>∇</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">y</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\nabla _{y}}$]]></tex-math></alternatives></inline-formula> are first-order finite-difference operators along the two spatial axes, and <inline-formula id="j_infor643_ineq_087"><alternatives><mml:math>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">target</mml:mtext>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[${\mathbf{X}_{b}^{\textit{target}}}$]]></tex-math></alternatives></inline-formula> is the target representation.</p>
<p>The frequency domain loss is defined as follows: 
<disp-formula id="j_infor643_eq_024">
<label>(24)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">freq</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mo largeop="true" movablelimits="false">∑</mml:mo></mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">B</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mo largeop="true" movablelimits="false">∑</mml:mo></mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">C</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mo largeop="true" movablelimits="false">∑</mml:mo></mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">l</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">L</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msubsup>
<mml:mrow>
<mml:mo fence="true" maxsize="1.19em" minsize="1.19em" stretchy="true">‖</mml:mo>
<mml:mi mathvariant="script">DWT</mml:mi>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">final</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>−</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">b</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo mathvariant="normal">,</mml:mo>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">target</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo fence="true" maxsize="1.19em" minsize="1.19em" stretchy="true">‖</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathcal{L}_{\textit{freq}}}={\sum \limits_{b=1}^{B}}{\sum \limits_{c=1}^{C}}{\sum \limits_{l=1}^{L}}{\big\| \mathcal{DWT}\big({\mathbf{X}_{b,c,l}^{\textit{final}}}-{\mathbf{X}_{b,c,l}^{\textit{target}}}\big)\big\| _{2}^{2}},\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_088"><alternatives><mml:math>
<mml:mi mathvariant="script">DWT</mml:mi></mml:math><tex-math><![CDATA[$\mathcal{DWT}$]]></tex-math></alternatives></inline-formula> represents the discrete wavelet transform, and <inline-formula id="j_infor643_ineq_089"><alternatives><mml:math>
<mml:mo stretchy="false">‖</mml:mo>
<mml:mo>·</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">‖</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup></mml:math><tex-math><![CDATA[$\| \cdot {\| _{2}^{2}}$]]></tex-math></alternatives></inline-formula> denotes the squared <inline-formula id="j_infor643_ineq_090"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi>ℓ</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\ell _{2}}$]]></tex-math></alternatives></inline-formula> norm. The correction here moves the DWT inside the L2 norm calculation: first compute the difference between the reconstructed and original image, then perform DWT on the difference tensor, and finally calculate the L2 norm of the transformed result to measure the reconstruction error in the frequency domain.</p>
<p>The total dual-domain loss is combined through adaptive weighting of the spatial and frequency domain losses: 
<disp-formula id="j_infor643_eq_025">
<label>(25)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">dual</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">ω</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">spatial</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">ω</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">freq</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathcal{L}_{\textit{dual}}}={\omega _{s}}{\mathcal{L}_{\textit{spatial}}}+{\omega _{f}}{\mathcal{L}_{\textit{freq}}},\]]]></tex-math></alternatives>
</disp-formula> 
where <inline-formula id="j_infor643_ineq_091"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">ω</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">s</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\omega _{s}}$]]></tex-math></alternatives></inline-formula> and <inline-formula id="j_infor643_ineq_092"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">ω</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">f</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\omega _{f}}$]]></tex-math></alternatives></inline-formula> are learnable weight parameters used to control the relative importance of the spatial and frequency domain losses. The final total loss is the weighted sum of the Focal Loss and the dual-domain loss: 
<disp-formula id="j_infor643_eq_026">
<label>(26)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">total</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">focal</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="italic">λ</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">dual</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[ {\mathcal{L}_{\textit{total}}}={\mathcal{L}_{\textit{focal}}}+\lambda {\mathcal{L}_{\textit{dual}}},\]]]></tex-math></alternatives>
</disp-formula> 
where <italic>λ</italic> is a learnable parameter that adjusts the weight balance between the focal loss and the dual-domain loss. By learning this weighted loss, the model can simultaneously optimize both classification accuracy and image reconstruction quality.</p>
</sec>
<sec id="j_infor643_s_017">
<label>3.6.2</label>
<title>Classification</title>
<p>After fusing cross-modal information through the Cross-Spectral Interaction Module, we project it into the classification space. The formulation is as follows: <disp-formula-group id="j_infor643_dg_001">
<disp-formula id="j_infor643_eq_027">
<label>(27)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true" columnalign="right left" columnspacing="0pt">
<mml:mtr>
<mml:mtd class="align-odd"/>
<mml:mtd class="align-even">
<mml:mi mathvariant="italic">r</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">(</mml:mo>
<mml:mo movablelimits="false">LN</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">fused</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo mathvariant="normal" fence="true" maxsize="1.19em" minsize="1.19em">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[\begin{aligned}{}& r={\mathbf{W}_{c}}\big(\operatorname{LN}({\mathbf{F}_{\textit{fused}}})\big)+{\mathbf{b}_{c}},\end{aligned}\]]]></tex-math></alternatives>
</disp-formula>
<disp-formula id="j_infor643_eq_028">
<label>(28)</label><alternatives><mml:math display="block">
<mml:mtable displaystyle="true" columnalign="right left" columnspacing="0pt">
<mml:mtr>
<mml:mtd class="align-odd"/>
<mml:mtd class="align-even">
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">ˆ</mml:mo></mml:mover>
<mml:mo>=</mml:mo>
<mml:mo movablelimits="false">softmax</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold">r</mml:mi>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo>
<mml:mo mathvariant="normal">,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable></mml:math><tex-math><![CDATA[\[\begin{aligned}{}& \hat{\mathbf{y}}=\operatorname{softmax}(\mathbf{r}),\end{aligned}\]]]></tex-math></alternatives>
</disp-formula>
</disp-formula-group> where <inline-formula id="j_infor643_ineq_093"><alternatives><mml:math>
<mml:mo movablelimits="false">LN</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">(</mml:mo>
<mml:mo>·</mml:mo>
<mml:mo mathvariant="normal" fence="true" stretchy="false">)</mml:mo></mml:math><tex-math><![CDATA[$\operatorname{LN}(\cdot )$]]></tex-math></alternatives></inline-formula> denotes layer normalization, <inline-formula id="j_infor643_ineq_094"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">K</mml:mi>
<mml:mo>×</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${\mathbf{W}_{c}}\in {\mathbb{R}^{K\times {d_{f}}}}$]]></tex-math></alternatives></inline-formula> and <inline-formula id="j_infor643_ineq_095"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">∈</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">K</mml:mi>
</mml:mrow>
</mml:msup></mml:math><tex-math><![CDATA[${\mathbf{b}_{c}}\in {\mathbb{R}^{K}}$]]></tex-math></alternatives></inline-formula> are classifier parameters distinct from the purification-gate parameters <inline-formula id="j_infor643_ineq_096"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">g</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\mathbf{W}_{g}}$]]></tex-math></alternatives></inline-formula> and <inline-formula id="j_infor643_ineq_097"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">g</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${\mathbf{b}_{g}}$]]></tex-math></alternatives></inline-formula>, <inline-formula id="j_infor643_ineq_098"><alternatives><mml:math>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="italic">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">f</mml:mi>
</mml:mrow>
</mml:msub></mml:math><tex-math><![CDATA[${d_{f}}$]]></tex-math></alternatives></inline-formula> is the fused feature dimension, and <italic>K</italic> is the number of classes (<inline-formula id="j_infor643_ineq_099"><alternatives><mml:math>
<mml:mi mathvariant="italic">K</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>3</mml:mn></mml:math><tex-math><![CDATA[$K=3$]]></tex-math></alternatives></inline-formula> for MVSA and <inline-formula id="j_infor643_ineq_100"><alternatives><mml:math>
<mml:mi mathvariant="italic">K</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>2</mml:mn></mml:math><tex-math><![CDATA[$K=2$]]></tex-math></alternatives></inline-formula> for HFM). The softmax output is consistent with the multiclass Focal Loss.</p>
</sec>
</sec>
</sec>
<sec id="j_infor643_s_018">
<label>4</label>
<title>Experimentation</title>
<sec id="j_infor643_s_019">
<label>4.1</label>
<title>Datasets</title>
<p>All our experiments were conducted on the publicly available datasets MVSA-Single, MVSA-Multiple (Niu <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_041">2016</xref>), and HFM (Cai <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_005">2019</xref>). MVSA-Multiple is an enhanced version of MVSA-Single, featuring a larger number of image-text pairs suitable for sentiment analysis. We adopted the same preprocessing strategy as Xu and Mao (<xref ref-type="bibr" rid="j_infor643_ref_070">2017</xref>). The HFM dataset is primarily used for multimodal sarcasm detection, a task with two sentiment polarities: positive and negative. For a fair comparison, we used the consistent data preprocessing method described in Cai <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_005">2019</xref>). A statistical analysis of the datasets is presented in Table <xref rid="j_infor643_tab_001">1</xref>.</p>
<p>MVSA-Single and MVSA-Multiple are available at.<xref ref-type="fn" rid="j_infor643_fn_001">1</xref><fn id="j_infor643_fn_001"><label><sup>1</sup></label>
<p><uri>https://mcrlab.net/research/mvsa-sentiment-analysis-on-multi-view-social-data/</uri></p></fn> The HFM dataset can be downloaded from.<xref ref-type="fn" rid="j_infor643_fn_002">2</xref><fn id="j_infor643_fn_002"><label><sup>2</sup></label>
<p><uri>https://github.com/headacheboy/data-of-multimodal-sarcasm-detection</uri></p></fn></p>
<table-wrap id="j_infor643_tab_001">
<label>Table 1</label>
<caption>
<p>Overview of data statistics.</p>
</caption>
<table>
<thead>
<tr>
<td style="vertical-align: top; text-align: left; border-top: solid thin; border-bottom: solid thin">Dataset</td>
<td style="vertical-align: top; text-align: left; border-top: solid thin; border-bottom: solid thin">Negative</td>
<td style="vertical-align: top; text-align: left; border-top: solid thin; border-bottom: solid thin">Neutral</td>
<td style="vertical-align: top; text-align: left; border-top: solid thin; border-bottom: solid thin">Positive</td>
<td style="vertical-align: top; text-align: left; border-top: solid thin; border-bottom: solid thin">Total</td>
</tr>
</thead>
<tbody>
<tr>
<td style="vertical-align: top; text-align: left">MVSA-S</td>
<td style="vertical-align: top; text-align: left">1358</td>
<td style="vertical-align: top; text-align: left">470</td>
<td style="vertical-align: top; text-align: left">2683</td>
<td style="vertical-align: top; text-align: left">4511</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">MVSA-M</td>
<td style="vertical-align: top; text-align: left">1298</td>
<td style="vertical-align: top; text-align: left">4408</td>
<td style="vertical-align: top; text-align: left">11318</td>
<td style="vertical-align: top; text-align: left">17024</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">HFM</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">14075</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">–</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">10560</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">24635</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="j_infor643_s_020">
<label>4.2</label>
<title>Hyperparameter Settings and Evaluation Metrics</title>
<sec id="j_infor643_s_021">
<label>4.2.1</label>
<title>Hyperparameters</title>
<p>The number of output categories is set according to the target dataset, i.e. 3 for MVSA-Single and MVSA-Multiple, and 2 for HFM, and the maximum text length is 100. The base learning rate is set to 1e-6 following validation-based tuning, while the GSSC module uses a larger learning rate of 3e-5. In the Focal Loss, the balancing factor <italic>α</italic> is set to a vector of all ones with a length of 3, and the focusing parameter <italic>γ</italic> is set to 2. The batch size is configured to 12, balancing training efficiency with memory usage. Furthermore, to enhance computational efficiency, avoid local optima, and achieve faster convergence, this paper utilizes the Adam optimizer. We train the model for 50 epochs in total. For the pre-trained model, we adopt RoBERTa-base (12 layers, hidden size 768) and ViT-B/16 (12 layers, patch size <inline-formula id="j_infor643_ineq_101"><alternatives><mml:math>
<mml:mn>16</mml:mn>
<mml:mo>×</mml:mo>
<mml:mn>16</mml:mn></mml:math><tex-math><![CDATA[$16\times 16$]]></tex-math></alternatives></inline-formula>) as the text and image encoders, respectively.</p>
</sec>
<sec id="j_infor643_s_022">
<label>4.2.2</label>
<title>Hardware Configuration</title>
<p>All experiments are conducted on a server equipped with an Intel(R) Xeon(R) Gold 6230 CPU @ 2.10 GHz, 256 GB RAM, and 1 NVIDIA RTX 4090 GPU with 24 GB memory. The runtime environment is Python 3.12, PyTorch 2.3.0, and CUDA 11.6. The average training time per epoch is approximately 12 minutes for the MVSA-Single dataset, 28 minutes for the MVSA-Multiple dataset, and 45 minutes for the HFM dataset.</p>
</sec>
<sec id="j_infor643_s_023">
<label>4.2.3</label>
<title>Evaluation Metrics</title>
<p>We adopt four widely used evaluation metrics from existing works: <inline-formula id="j_infor643_ineq_102"><alternatives><mml:math>
<mml:mtext mathvariant="italic">Accuracy</mml:mtext></mml:math><tex-math><![CDATA[$\textit{Accuracy}$]]></tex-math></alternatives></inline-formula>, <inline-formula id="j_infor643_ineq_103"><alternatives><mml:math>
<mml:mtext mathvariant="italic">Precision</mml:mtext></mml:math><tex-math><![CDATA[$\textit{Precision}$]]></tex-math></alternatives></inline-formula>, <inline-formula id="j_infor643_ineq_104"><alternatives><mml:math>
<mml:mtext mathvariant="italic">Recall</mml:mtext></mml:math><tex-math><![CDATA[$\textit{Recall}$]]></tex-math></alternatives></inline-formula>, and <inline-formula id="j_infor643_ineq_105"><alternatives><mml:math>
<mml:mtext mathvariant="italic">F1-score</mml:mtext></mml:math><tex-math><![CDATA[$\textit{F1-score}$]]></tex-math></alternatives></inline-formula>.</p>
</sec>
</sec>
<sec id="j_infor643_s_024">
<label>4.3</label>
<title>Comparison of Experimental Results</title>
<sec id="j_infor643_s_025">
<label>4.3.1</label>
<title>Baselines</title>
<p>To validate the feasibility of our approach, we compare it with various existing methods, detailed as follows:</p>
<p><italic><bold>Text-modality mode.</bold></italic></p>
<list>
<list-item id="j_infor643_li_004">
<label>•</label>
<p>CNN (Kim, <xref ref-type="bibr" rid="j_infor643_ref_029">2014</xref>): A Convolutional Neural Network that achieves its results by repeatedly convolving over local features, making it suitable for scenarios involving local pattern modelling.</p>
</list-item>
<list-item id="j_infor643_li_005">
<label>•</label>
<p>BiLSTM (Zhou <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_084">2016</xref>): A variant of RNN that has achieved excellent performance in text processing through its sophisticated gating mechanisms.</p>
</list-item>
<list-item id="j_infor643_li_006">
<label>•</label>
<p>BERT (Devlin <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_012">2019</xref>): A pre-trained model learned from a large corpus of knowledge, offering stronger feature representation capabilities compared to traditional neural networks.</p>
</list-item>
</list>
<p><italic><bold>Image-modality mode.</bold></italic></p>
<list>
<list-item id="j_infor643_li_007">
<label>•</label>
<p>ResNet-50 (He <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_022">2016</xref>): A pre-trained model based on a Convolutional Neural Network architecture.</p>
</list-item>
<list-item id="j_infor643_li_008">
<label>•</label>
<p>ViT (Dosovitskiy <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_014">2020</xref>): A Transformer-based model that processes images by treating them as sequences.</p>
</list-item>
</list>
<p><italic><bold>Multi-Modal mode.</bold></italic> 
<list>
<list-item id="j_infor643_li_009">
<label>•</label>
<p>MultiSentiNet (Xu and Mao, <xref ref-type="bibr" rid="j_infor643_ref_070">2017</xref>): Identify sentiment by extracting vocabulary that significantly influences the sentiment of the entire tweet.</p>
</list-item>
<list-item id="j_infor643_li_010">
<label>•</label>
<p>HSAN (Xu, <xref ref-type="bibr" rid="j_infor643_ref_069">2017</xref>): The model leverages image captions to derive visual features, providing extra context for the text.</p>
</list-item>
<list-item id="j_infor643_li_011">
<label>•</label>
<p>Co-MN-Hop6 (Xu <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_071">2018</xref>): Modelling the interaction between image and text and using shared memory to assist in the analysis.</p>
</list-item>
<list-item id="j_infor643_li_012">
<label>•</label>
<p>MGNNS (Yang <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_073">2021</xref>): A multi-channel graph neural network has been designed specifically for sentiment detection.</p>
</list-item>
<list-item id="j_infor643_li_013">
<label>•</label>
<p>CLMLF (Li <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_033">2022</xref>): It aligns and fuses multimodal features based on a Transformer encoder, and employs two contrastive learning tasks to assist in modelling.</p>
</list-item>
<list-item id="j_infor643_li_014">
<label>•</label>
<p>ITIN (Zhu <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_090">2022</xref>): By utilizing cross-modal alignment and gating mechanisms, the performance of the model has been improved.</p>
</list-item>
<list-item id="j_infor643_li_015">
<label>•</label>
<p>MVCN (Wei <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_066">2023</xref>): A new Multi-View Calibration Network designed to tackle the issues of modality heterogeneity.</p>
</list-item>
<list-item id="j_infor643_li_016">
<label>•</label>
<p>CTMWA (Zhang <italic>et al.</italic>, <xref ref-type="bibr" rid="j_infor643_ref_076">2024</xref>): The reliability of the model has been enhanced in cases of modality absence and information imbalance through a cross-modal translation network and a single-modal weight adaptation strategy.</p>
</list-item>
</list>
</p>
</sec>
<sec id="j_infor643_s_026">
<label>4.3.2</label>
<title>Comparative Experimental Results and Analysis</title>
<table-wrap id="j_infor643_tab_002">
<label>Table 2</label>
<caption>
<p>Experimental results of multiple comparative models on MVSA and HFM datasets.</p>
</caption>
<table>
<thead>
<tr>
<td rowspan="2" style="vertical-align: middle; text-align: left; border-top: solid thin; border-bottom: solid thin">Modality</td>
<td rowspan="2" style="vertical-align: middle; text-align: left; border-top: solid thin; border-bottom: solid thin">Method</td>
<td colspan="2" style="vertical-align: top; text-align: left; border-top: solid thin; border-bottom: solid thin">MVSA-single</td>
<td colspan="2" style="vertical-align: top; text-align: left; border-top: solid thin; border-bottom: solid thin">MVSA-multiple</td>
<td colspan="2" style="vertical-align: top; text-align: left; border-top: solid thin; border-bottom: solid thin">HFM</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">Acc.</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">F1.</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">Acc.</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">F1.</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">Acc.</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">F1.</td>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="3" style="vertical-align: middle; text-align: left">Text</td>
<td style="vertical-align: top; text-align: left">CNN</td>
<td style="vertical-align: top; text-align: left">0.6819</td>
<td style="vertical-align: top; text-align: left">0.559</td>
<td style="vertical-align: top; text-align: left">0.6564</td>
<td style="vertical-align: top; text-align: left">0.5766</td>
<td style="vertical-align: top; text-align: left">0.8003</td>
<td style="vertical-align: top; text-align: left">0.7532</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">BiLSTM</td>
<td style="vertical-align: top; text-align: left">0.7012</td>
<td style="vertical-align: top; text-align: left">0.6506</td>
<td style="vertical-align: top; text-align: left">0.679</td>
<td style="vertical-align: top; text-align: left">0.679</td>
<td style="vertical-align: top; text-align: left">0.819</td>
<td style="vertical-align: top; text-align: left">0.7753</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">BERT</td>
<td style="vertical-align: top; text-align: left">0.7111</td>
<td style="vertical-align: top; text-align: left">0.697</td>
<td style="vertical-align: top; text-align: left">0.6759</td>
<td style="vertical-align: top; text-align: left">0.6624</td>
<td style="vertical-align: top; text-align: left">0.8389</td>
<td style="vertical-align: top; text-align: left">0.8326</td>
</tr>
<tr>
<td rowspan="2" style="vertical-align: middle; text-align: left">Image</td>
<td style="vertical-align: top; text-align: left">ResNet-50</td>
<td style="vertical-align: top; text-align: left">0.6467</td>
<td style="vertical-align: top; text-align: left">0.6155</td>
<td style="vertical-align: top; text-align: left">0.6188</td>
<td style="vertical-align: top; text-align: left">0.6098</td>
<td style="vertical-align: top; text-align: left">0.7277</td>
<td style="vertical-align: top; text-align: left">0.7138</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">ViT</td>
<td style="vertical-align: top; text-align: left">0.6378</td>
<td style="vertical-align: top; text-align: left">0.6226</td>
<td style="vertical-align: top; text-align: left">0.6194</td>
<td style="vertical-align: top; text-align: left">0.6119</td>
<td style="vertical-align: top; text-align: left">0.7309</td>
<td style="vertical-align: top; text-align: left">0.7152</td>
</tr>
<tr>
<td rowspan="8" style="vertical-align: middle; text-align: left; border-bottom: solid thin">Multimodal</td>
<td style="vertical-align: top; text-align: left">MultiSentiNet</td>
<td style="vertical-align: top; text-align: left">0.6984</td>
<td style="vertical-align: top; text-align: left">0.6984</td>
<td style="vertical-align: top; text-align: left">0.6886</td>
<td style="vertical-align: top; text-align: left">0.6811</td>
<td style="vertical-align: top; text-align: left">0.8103</td>
<td style="vertical-align: top; text-align: left">0.7799</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">HSAN</td>
<td style="vertical-align: top; text-align: left">0.6988</td>
<td style="vertical-align: top; text-align: left">0.669</td>
<td style="vertical-align: top; text-align: left">0.6796</td>
<td style="vertical-align: top; text-align: left">0.6776</td>
<td style="vertical-align: top; text-align: left">0.8174</td>
<td style="vertical-align: top; text-align: left">0.7874</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">Co-MN-Hop6</td>
<td style="vertical-align: top; text-align: left">0.7051</td>
<td style="vertical-align: top; text-align: left">0.7001</td>
<td style="vertical-align: top; text-align: left">0.6892</td>
<td style="vertical-align: top; text-align: left">0.6883</td>
<td style="vertical-align: top; text-align: left">0.8344</td>
<td style="vertical-align: top; text-align: left">0.8018</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">MGNNS</td>
<td style="vertical-align: top; text-align: left">0.7377</td>
<td style="vertical-align: top; text-align: left">0.727</td>
<td style="vertical-align: top; text-align: left">0.7249</td>
<td style="vertical-align: top; text-align: left">0.6934</td>
<td style="vertical-align: top; text-align: left">0.8402</td>
<td style="vertical-align: top; text-align: left">0.8060</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">CLMLF</td>
<td style="vertical-align: top; text-align: left">0.7533</td>
<td style="vertical-align: top; text-align: left">0.7346</td>
<td style="vertical-align: top; text-align: left">0.7200</td>
<td style="vertical-align: top; text-align: left">0.6983</td>
<td style="vertical-align: top; text-align: left">0.8543</td>
<td style="vertical-align: top; text-align: left">0.8487</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">MVCN</td>
<td style="vertical-align: top; text-align: left">0.7606</td>
<td style="vertical-align: top; text-align: left">0.7455</td>
<td style="vertical-align: top; text-align: left">0.7207</td>
<td style="vertical-align: top; text-align: left">0.7001</td>
<td style="vertical-align: top; text-align: left">0.8568</td>
<td style="vertical-align: top; text-align: left">0.8523</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">CTMWA</td>
<td style="vertical-align: top; text-align: left">0.7591</td>
<td style="vertical-align: top; text-align: left">0.7574</td>
<td style="vertical-align: top; text-align: left">0.7402</td>
<td style="vertical-align: top; text-align: left">0.7384</td>
<td style="vertical-align: top; text-align: left">–</td>
<td style="vertical-align: top; text-align: left">–</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">Ours</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin"><bold>0.7711</bold></td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin"><bold>0.7680</bold></td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin"><bold>0.7406</bold></td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin"><bold>0.7400</bold></td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin"><bold>0.8589</bold></td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin"><bold>0.8405</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To effectively assess our model’s performance, we conducted extensive comparative experiments, as shown in Table <xref rid="j_infor643_tab_002">2</xref>. Additionally, we performed the following experimental analyses:</p>
<p><bold>1) Experimental data consistently show that the text modality generally outperforms the image modality in all tests</bold>. Specifically, the BERT model outperformed the image-based ViT and ResNet-50 models. For instance, in the MVSA-Single task, BERT achieved an accuracy of 0.7111 and an F1-score of 0.6970, whereas ViT’s accuracy was only 0.6378 and its F1-score was 0.6226, significantly lower than BERT’s. The superior performance of the text modality is primarily attributed to the rich contextual information inherent in text data, which is crucial for sentiment analysis. BERT can capture deeper semantic associations and sentiment features, thus exhibiting stronger expressive capabilities in sentiment analysis tasks. In contrast, sentiment features in the image modality are often more implicit, and extracting image features is influenced by factors such as content and resolution, which can lead to inferior performance in sentiment analysis. Therefore, the powerful semantic modelling capability of the text modality provides a more distinct advantage in sentiment analysis tasks, while the image modality performs relatively weaker due to its limitations.</p>
<p><bold>2) Compared to single-modality approaches, multimodal methods generally exhibit superior performance</bold>. For example, in the MVSA-Single task, the multimodal method CTMWA achieved an accuracy of 0.7591 and an F1-score of 0.7574, while the single-text modality BERT had an accuracy of 0.7111 and an F1-score of 0.6970, and the single-image modality ResNet-50 performed even worse. Multimodal methods leverage the strengths of both images and text by integrating them. Text provides rich semantic information, while an image can offer complementary visual cues, especially for sentiment with visual manifestations. For instance, in tasks with clear sentiment polarity, an image might help the system understand the sentiment context, while text provides precise sentiment words and contextual information. The complementarity of the two modalities enables multimodal methods to provide more comprehensive information in sentiment analysis tasks, thereby improving the model’s accuracy and robustness. Consequently, multimodal methods effectively enhance performance in sentiment analysis tasks through cross-modal feature fusion, capturing sentiment information more comprehensively than single modalities and thus improving overall performance.</p>
<p><bold>3) FDSF-Net achieves competitive, but not uniformly superior, performance relative to the compared multimodal methods</bold>. In Table <xref rid="j_infor643_tab_002">2</xref>, it obtains the highest reported accuracy and F1-score on MVSA-Single and the highest reported values on MVSA-Multiple by small margins under the adopted single-run setting. On HFM, it achieves the highest accuracy (0.8589), whereas its F1-score (0.8405) is lower than those of CLMLF (0.8487) and MVCN (0.8523). Accordingly, these results support the competitiveness of the proposed design, but they do not establish uniform superiority. Because repeated-run variance and significance tests are unavailable, the small numerical differences, particularly on MVSA-Multiple, should be interpreted cautiously and not as statistically significant improvements.</p>
<p>The HFM results further suggest that FDSF-Net can be applied to multimodal sarcasm detection. Its accuracy is competitive with the compared methods, while its F1-score does not exceed the strongest baselines. We therefore regard this experiment as preliminary evidence of cross-task applicability rather than proof of superior generalization.</p>
</sec>
<sec id="j_infor643_s_027">
<label>4.3.3</label>
<title>Ablation Study Results and Analysis</title>
<p>To thoroughly validate the effectiveness of our proposed method, we conducted ablation experiments on the two MVSA datasets and HFM, as shown in Table <xref rid="j_infor643_tab_003">3</xref>. The experimental scenarios are primarily as follows:</p>
<list>
<list-item id="j_infor643_li_017">
<label>•</label>
<p>Only I: Indicates using only image representation.</p>
</list-item>
<list-item id="j_infor643_li_018">
<label>•</label>
<p>Only T: Indicates using only text representation.</p>
</list-item>
<list-item id="j_infor643_li_019">
<label>•</label>
<p>w/o DFDD: Represents removing the Dynamic Frequency Domain Decoupling module to observe its impact on model performance.</p>
</list-item>
<list-item id="j_infor643_li_020">
<label>•</label>
<p>w/o DFDD-H: Removes the image HF-domain operation.</p>
</list-item>
<list-item id="j_infor643_li_021">
<label>•</label>
<p>w/o DFDD-L: Removes the image LF-domain processing operation.</p>
</list-item>
<list-item id="j_infor643_li_022">
<label>•</label>
<p>w/o DFDD-D: Removes both the image LF- and HF-domain operations, retaining only the frequency transformation.</p>
</list-item>
<list-item id="j_infor643_li_023">
<label>•</label>
<p>w/o C: Removes the Cross-Spectral Interaction Module.</p>
</list-item>
<list-item id="j_infor643_li_024">
<label>•</label>
<p>w/o M: Removes the Multi-Granularity Semantic Purification module.</p>
</list-item>
<list-item id="j_infor643_li_025">
<label>•</label>
<p>w/o S: Removes the Spectral Domain Dual Domain Loss, retaining only the classification loss.</p>
</list-item>
<list-item id="j_infor643_li_026">
<label>•</label>
<p>All: Our proposed complete model.</p>
</list-item>
</list>
<table-wrap id="j_infor643_tab_003">
<label>Table 3</label>
<caption>
<p>Ablation study results, MFS denotes equally fusing the image and text features as Li <italic>et al.</italic> (<xref ref-type="bibr" rid="j_infor643_ref_033">2022</xref>).</p>
</caption>
<table>
<thead>
<tr>
<td rowspan="2" style="vertical-align: middle; text-align: left; border-top: solid thin; border-bottom: solid thin">Model</td>
<td colspan="2" style="vertical-align: top; text-align: left; border-top: solid thin; border-bottom: solid thin">MVSA-single</td>
<td colspan="2" style="vertical-align: top; text-align: left; border-top: solid thin; border-bottom: solid thin">MVSA-multiple</td>
<td colspan="2" style="vertical-align: top; text-align: left; border-top: solid thin; border-bottom: solid thin">HFM</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">Acc.</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">F1.</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">Acc.</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">F1.</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">Acc.</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">F1.</td>
</tr>
</thead>
<tbody>
<tr>
<td style="vertical-align: top; text-align: left">BERT</td>
<td style="vertical-align: top; text-align: left">0.7111</td>
<td style="vertical-align: top; text-align: left">0.697</td>
<td style="vertical-align: top; text-align: left">0.6759</td>
<td style="vertical-align: top; text-align: left">0.6624</td>
<td style="vertical-align: top; text-align: left">0.8389</td>
<td style="vertical-align: top; text-align: left">0.8326</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">ViT</td>
<td style="vertical-align: top; text-align: left">0.6378</td>
<td style="vertical-align: top; text-align: left">0.6226</td>
<td style="vertical-align: top; text-align: left">0.6194</td>
<td style="vertical-align: top; text-align: left">0.6119</td>
<td style="vertical-align: top; text-align: left">0.7309</td>
<td style="vertical-align: top; text-align: left">0.7152</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">MFS</td>
<td style="vertical-align: top; text-align: left">0.7217</td>
<td style="vertical-align: top; text-align: left">0.7205</td>
<td style="vertical-align: top; text-align: left">0.7063</td>
<td style="vertical-align: top; text-align: left">0.6851</td>
<td style="vertical-align: top; text-align: left">0.8434</td>
<td style="vertical-align: top; text-align: left">0.8375</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">Only I</td>
<td style="vertical-align: top; text-align: left">0.6988</td>
<td style="vertical-align: top; text-align: left">0.669</td>
<td style="vertical-align: top; text-align: left">0.6796</td>
<td style="vertical-align: top; text-align: left">0.6776</td>
<td style="vertical-align: top; text-align: left">0.7309</td>
<td style="vertical-align: top; text-align: left">0.7152</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">Only T</td>
<td style="vertical-align: top; text-align: left">0.7051</td>
<td style="vertical-align: top; text-align: left">0.7001</td>
<td style="vertical-align: top; text-align: left">0.6892</td>
<td style="vertical-align: top; text-align: left">0.6883</td>
<td style="vertical-align: top; text-align: left">–</td>
<td style="vertical-align: top; text-align: left">–</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">w/o DS</td>
<td style="vertical-align: top; text-align: left">0.7511</td>
<td style="vertical-align: top; text-align: left">0.7458</td>
<td style="vertical-align: top; text-align: left">0.7294</td>
<td style="vertical-align: top; text-align: left">0.7284</td>
<td style="vertical-align: top; text-align: left">0.8468</td>
<td style="vertical-align: top; text-align: left">0.8286</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">w/o DS-H</td>
<td style="vertical-align: top; text-align: left">0.7644</td>
<td style="vertical-align: top; text-align: left">0.7590</td>
<td style="vertical-align: top; text-align: left">0.7329</td>
<td style="vertical-align: top; text-align: left">0.7319</td>
<td style="vertical-align: top; text-align: left">0.8522</td>
<td style="vertical-align: top; text-align: left">0.8339</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">w/o DS-L</td>
<td style="vertical-align: top; text-align: left">0.7578</td>
<td style="vertical-align: top; text-align: left">0.7524</td>
<td style="vertical-align: top; text-align: left">0.7318</td>
<td style="vertical-align: top; text-align: left">0.7308</td>
<td style="vertical-align: top; text-align: left">0.8506</td>
<td style="vertical-align: top; text-align: left">0.8323</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">w/o DS-D</td>
<td style="vertical-align: top; text-align: left">0.7556</td>
<td style="vertical-align: top; text-align: left">0.7502</td>
<td style="vertical-align: top; text-align: left">0.7312</td>
<td style="vertical-align: top; text-align: left">0.7302</td>
<td style="vertical-align: top; text-align: left">0.8489</td>
<td style="vertical-align: top; text-align: left">0.8307</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">w/o C</td>
<td style="vertical-align: top; text-align: left">0.7622</td>
<td style="vertical-align: top; text-align: left">0.7568</td>
<td style="vertical-align: top; text-align: left">0.7371</td>
<td style="vertical-align: top; text-align: left">0.7361</td>
<td style="vertical-align: top; text-align: left">0.8535</td>
<td style="vertical-align: top; text-align: left">0.8352</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">w/o M</td>
<td style="vertical-align: top; text-align: left">0.7667</td>
<td style="vertical-align: top; text-align: left">0.7613</td>
<td style="vertical-align: top; text-align: left">0.7388</td>
<td style="vertical-align: top; text-align: left">0.7378</td>
<td style="vertical-align: top; text-align: left">0.8551</td>
<td style="vertical-align: top; text-align: left">0.8367</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left">w/o S</td>
<td style="vertical-align: top; text-align: left">0.7689</td>
<td style="vertical-align: top; text-align: left">0.7634</td>
<td style="vertical-align: top; text-align: left">0.7400</td>
<td style="vertical-align: top; text-align: left">0.7390</td>
<td style="vertical-align: top; text-align: left">0.8572</td>
<td style="vertical-align: top; text-align: left">0.8388</td>
</tr>
<tr>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin">All</td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin"><bold>0.7711</bold></td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin"><bold>0.7680</bold></td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin"><bold>0.7406</bold></td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin"><bold>0.7400</bold></td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin"><bold>0.8589</bold></td>
<td style="vertical-align: top; text-align: left; border-bottom: solid thin"><bold>0.8405</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The ablation results show the largest observed decrease when the Dynamic Frequency Domain Decoupling (DFDD) module is removed, followed by the DFDD-D, DFDD-L, and DFDD-H variants. Removing the complete DFDD module eliminates differentiated LF/HF processing; removing DFDD-D retains the transform but removes the decoupling operations; and the DFDD-L and DFDD-H variants isolate the contributions of low- and high-frequency processing. These are descriptive single-run comparisons and are not presented as statistically significant differences.</p>
<p>Furthermore, we analysed the contributions of the Multi-Granularity Semantic Purification Module (M), Cross-Spectral Interaction Module (C), and Spectral Domain Dual Domain Loss (S) to model performance. The experimental results show that removing the M module had the most significant impact on model performance, followed by the C and S modules. Specifically, after removing “w/o M”, the model lost its ability to perform multi-granularity purification of text information. This module optimizes text features via multi-level semantic extraction, enabling the model to better understand the complex relationship between images and text. Without the M module, the model’s accuracy in multimodal understanding significantly decreased, particularly in deep fusion of image and text. After removing “w/o C”, the cross-spectral interaction module no longer established an effective connection between text and image features, leading to a noticeable weakening of their fusion capability and affecting the model’s overall performance in cross-modal tasks. Finally, after removing “w/o S”, although this loss function helps enhance the consistency between image and text in the frequency domain, allowing them to align better, its impact on overall performance improvement was relatively limited compared to other modules. Overall, despite the unique role of the S module, its contribution to performance improvement was less significant than that of the M and C modules, indicating that the precise extraction and efficient fusion of text and image features are crucial for the success of multimodal tasks.</p>
<p>In summary, the ablation results indicate that each module contributes to the reported single-run performance. In particular, removing the Dynamic Frequency Domain Decoupling module produces the largest observed decrease. These descriptive differences support the usefulness, rather than establish the statistical superiority, of the complete model.</p>
<fig id="j_infor643_fig_003">
<label>Fig. 3</label>
<caption>
<p>Wavelet order analysis.</p>
</caption>
<graphic xlink:href="infor643_g003.jpg"/>
</fig>
</sec>
</sec>
<sec id="j_infor643_s_028">
<label>4.4</label>
<title>Hyperparameter Analysis</title>
<p>In deep learning, the choice of hyperparameters significantly impacts model performance. For the Multimodal Sentiment Analysis Model with Frequency-Domain Decoupling and Semantic Filtering proposed in this paper, the order of the Discrete Wavelet Transform (DWT) is a crucial hyperparameter. To thoroughly understand its influence on model performance, a series of experimental analyses was conducted, as shown in Fig. <xref rid="j_infor643_fig_003">3</xref>. The results indicate that the model performs best when using the db2 wavelet basis, while db1 performs significantly worse, and the performance of db3, db4, and db5 decreases sequentially.</p>
<p>Specifically, the advantage of the db2 wavelet basis lies primarily in its balanced frequency-domain decoupling, which enables it to effectively process both LF global semantic information and HF noise in an image. db2 can not only precisely extract important semantic features from an image but also avoid excessive filtering that could lead to the loss of sentiment information, thereby enhancing overall model performance. In contrast, while the db1 wavelet basis is relatively simpler, its frequency-domain decomposition ability is weaker, failing to effectively handle complex image details. Our method provided insufficient noise suppression, leading to poorer model performance. Furthermore, although db3, db4, and db5 may offer advantages in processing certain frequency-domain details, their overly detailed decomposition methods can lead to excessive filtering and loss of image information, especially in finding a suitable balance between noise reduction and feature retention, ultimately causing a gradual decline in model performance. Therefore, db2 performs optimally in this model because it can suppress unnecessary noise while preserving effective information, thus maximizing the model’s effectiveness.</p>
</sec>
<sec id="j_infor643_s_029">
<label>4.5</label>
<title>Visual Analysis</title>
<p>To gain a deeper, more intuitive understanding of how our model processes multimodal data, we employed various visualization methods to observe it from different perspectives.</p>
<fig id="j_infor643_fig_004">
<label>Fig. 4</label>
<caption>
<p>Sample heatmap of MVSA dataset.</p>
</caption>
<graphic xlink:href="infor643_g004.jpg"/>
</fig>
<sec id="j_infor643_s_030">
<label>4.5.1</label>
<title>Heatmap Analysis</title>
<p>To further explore the sentiment information our proposed model learns from multimodal data, we conducted a visual analysis of four sample sets from the MVSA dataset using Grad-CAM. We computed each token’s contribution to the classification result using the model’s last attention layer’s output and gradients. These contributions were then mapped back to the original image’s spatial locations, generating heatmaps as shown in Fig. <xref rid="j_infor643_fig_004">4</xref>.</p>
<p>The heatmaps reveal that when key semantic features in the text are taken into account, critical regions of the image receive more attention. Based on the model’s capture of the image’s global structure, irrelevant information did not hinder attention to these crucial regions. This is primarily thanks to the model’s understanding of the image’s overall structure, and also due to our method’s ability to, to some extent, mitigate the interference of potential noise in the image on the model’s comprehension. For example, as shown in the first example of Fig. <xref rid="j_infor643_fig_004">4</xref>(a), when keywords like “PresidentRaps” and “singing” appear in the text, the model focuses more on the person’s area in the image, making it easier to extract sentiment cues and enhance the understanding of user sentiment. When considering the potential sentiment information contained in “treat them” in Fig. <xref rid="j_infor643_fig_004">4</xref>(d), the “them” in the image was effectively attended to by the model, which is attributed to our model’s comprehensive understanding of the image’s global structure.</p>
</sec>
<sec id="j_infor643_s_031">
<label>4.5.2</label>
<title>Feature Distribution Analysis</title>
<p>To more intuitively demonstrate the spatial distribution of features before and after processing by the Dynamic Frequency Domain Decoupling Module, we used t-SNE to reduce the dimensionality of features from the MVSA dataset and visualized the results as 2D scatter plots, as shown in Fig. <xref rid="j_infor643_fig_005">5</xref>.</p>
<p>By comparing the visualization effects of both scenarios, we can clearly observe that before processing by the Dynamic Frequency Domain Decoupling Module, features of the same class were relatively dispersed in space, with significant overlap between different classes. This sparsity in feature distribution and inter-class overlap made it difficult for the model to distinguish among categories, resulting in lower performance. However, after processing by the Dynamic Frequency Domain Decoupling Module, we observed that features of the same class tended to cluster spatially, and the overlap between classes decreased significantly. This optimized feature distribution provided the model with clearer class boundaries, enabling it to more easily identify and distinguish different sentiment categories, thereby significantly improving model performance in sentiment analysis tasks.</p>
<fig id="j_infor643_fig_005">
<label>Fig. 5</label>
<caption>
<p>The feature space distribution of the MVSA dataset before and after processing by the DSDD module.</p>
</caption>
<graphic xlink:href="infor643_g005.jpg"/>
</fig>
</sec>
<sec id="j_infor643_s_032">
<label>4.5.3</label>
<title>Visual Analysis of Different Sub Bands</title>
<fig id="j_infor643_fig_006">
<label>Fig. 6</label>
<caption>
<p>PCA dimensionality reduction visualization of image frequency domain transformation process.</p>
</caption>
<graphic xlink:href="infor643_g006.jpg"/>
</fig>
<p>In the frequency-domain analysis of an image, LF components carry the main content and global structural information, typically reflecting its overall outline, basic shapes, and general features. These LF components are crucial to an image’s overall perception, as they convey its primary structure and visual characteristics. In the frequency domain, the LF region is often the key area that best represents the image’s global structure. Therefore, by extracting LF components, we can intuitively obtain a global view of the image.</p>
<p>As shown in Fig. <xref rid="j_infor643_fig_006">6</xref>, as the frequency gradually increases, the HF components of the image begin to emerge. This HF information contains the detailed parts of the image, such as textures, edges, and minute structural features. The HF provides finer image details, making the contours and layers of these details clearer. However, an increase in frequency not only provides more detail but also introduces additional noise components, which typically manifest as chaotic HF variations in the image. As the frequency increases, the global structural information in the image gradually blurs, while details become more complex and contain more irrelevant noise. Therefore, in image processing, balancing the extraction of LF and HF information, especially effective noise suppression, is a critical issue in image analysis. Through frequency domain analysis, we can clearly distinguish the main structural components from noise in an image, thereby enhancing the model’s sentiment analysis capabilities.</p>
</sec>
</sec>
<sec id="j_infor643_s_033">
<label>4.6</label>
<title>Case Study</title>
<p>To further explore the performance of our proposed model in sentiment analysis, we selected several representative cases for in-depth analysis, as shown in Fig. <xref rid="j_infor643_fig_007">7</xref>. This aims to comprehensively evaluate the model’s robustness and accuracy. The analysis results indicate that when image information is unclear or somewhat ambiguous, the model struggles to determine sentiment polarity, primarily because it lacks the ability to extract and understand the global sentiment tone. For instance, in the first case, despite explicit sentiment cues such as “feeling dizzy lol” in the text, the model’s confidence in its judgment was low due to the ambiguity of the image’s sentiment.</p>
<p>In contrast, in the other two cases in the table, the model demonstrated higher confidence in its judgment. For example, in the third case, the prominent flowers in the image mostly convey positive sentiment. Concurrently, the model effectively suppresses noise information outside the target object in the image, preventing it from interfering with sentiment analysis. The results above indicate that our proposed model has strong capabilities for extracting sentiment information from both images and text, thereby enhancing its overall performance in sentiment analysis.</p>
<fig id="j_infor643_fig_007">
<label>Fig. 7</label>
<caption>
<p>Case study.</p>
</caption>
<graphic xlink:href="infor643_g007.jpg"/>
</fig>
</sec>
</sec>
<sec id="j_infor643_s_034">
<label>5</label>
<title>Concluding Remarks and Future Work</title>
<p>This work introduces the Frequency Domain Decoupling and Semantic Filtering Network (FDSF-Net) for multimodal sentiment analysis. The model addresses multimodal noise through differentiated processing of image-frequency components and semantic filtering of text. Under the reported single-run setting, FDSF-Net achieves competitive results on the evaluated sentiment-analysis datasets and shows potential applicability to multimodal sarcasm detection. These results should not be interpreted as statistically significant differences because repeated-run variance estimates are not available.</p>
<p>However, our method has limitations. When the sentiment information in an image is unclear or ambiguous, the model struggles to determine sentiment polarity. Its discriminative performance is particularly limited when the global sentiment tone is indistinct or when there is significant interfering information in the image. This might be due to our model’s insufficient handling of global sentiment tone information. Future work will focus on more refined image processing methods. To address the model’s limitations when image sentiment is unclear, we plan to incorporate more advanced visual understanding techniques to improve its sensitivity to subtle sentiment expressions in images. Additionally, future efforts could involve introducing additional sentiment-related multimodal data and optimizing data preprocessing methods to further improve the model’s robustness and accuracy across complex, varied scenarios.</p>
</sec>
</body>
<back>
<ref-list id="j_infor643_reflist_001">
<title>References</title>
<ref id="j_infor643_ref_001">
<mixed-citation publication-type="journal"><string-name><surname>An</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Zainon</surname>, <given-names>W.M.N.W.</given-names></string-name> (<year>2023</year>). <article-title>Integrating color cues to improve multimodal sentiment analysis in social media</article-title>. <source>Engineering Applications of Artificial Intelligence</source>, <volume>126</volume>, <elocation-id>106874</elocation-id>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_002">
<mixed-citation publication-type="journal"><string-name><surname>Aziz</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Chowdhury</surname>, <given-names>N.K.</given-names></string-name>, <string-name><surname>Kabir</surname>, <given-names>M.A.</given-names></string-name>, <string-name><surname>Chy</surname>, <given-names>A.N.</given-names></string-name>, <string-name><surname>Siddique</surname>, <given-names>Md.J.</given-names></string-name> (<year>2025</year>). <article-title>MMTF-DES: a fusion of multimodal transformer models for desire, emotion, and sentiment analysis of social media data</article-title>. <source>Neurocomputing</source>, <volume>623</volume>, <elocation-id>129376</elocation-id>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_003">
<mixed-citation publication-type="other"><string-name><surname>Ba</surname>, <given-names>J.L.</given-names></string-name>, <string-name><surname>Kiros</surname>, <given-names>J.R.</given-names></string-name>, <string-name><surname>Hinton</surname>, <given-names>G.E.</given-names></string-name> (2016). Layer Normalization. arXiv preprint arXiv:<ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1607.06450">1607.06450</ext-link>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_004">
<mixed-citation publication-type="journal"><string-name><surname>Bibi</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Abbasi</surname>, <given-names>W.A.</given-names></string-name>, <string-name><surname>Aziz</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Khalil</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Uddin</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Iwendi</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Gadekallu</surname>, <given-names>T.R.</given-names></string-name> (<year>2022</year>). <article-title>A novel unsupervised ensemble framework using concept-based linguistic methods and machine learning for twitter sentiment analysis</article-title>. <source>Pattern Recognition Letters</source>, <volume>158</volume>, <fpage>80</fpage>–<lpage>86</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_005">
<mixed-citation publication-type="chapter"><string-name><surname>Cai</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Cai</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Wan</surname>, <given-names>X.</given-names></string-name> (<year>2019</year>). <chapter-title>Multi-modal sarcasm detection in twitter with hierarchical fusion model</chapter-title>. In: <source>Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics</source>, pp. <fpage>2506</fpage>–<lpage>2515</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_006">
<mixed-citation publication-type="journal"><string-name><surname>Chen</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Su</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Wu</surname>, <given-names>P.</given-names></string-name>, <string-name><surname>Hua</surname>, <given-names>B.</given-names></string-name> (<year>2023</year>). <article-title>Joint multimodal sentiment analysis based on information relevance</article-title>. <source>Information Processing &amp; Management</source>, <volume>60</volume>(<issue>2</issue>), <elocation-id>103193</elocation-id>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_007">
<mixed-citation publication-type="other"><string-name><surname>Chen</surname>, <given-names>Y.</given-names></string-name> (2015). <italic>Convolutional Neural Network for Sentence Classification</italic>. Master’s thesis, University of Waterloo.</mixed-citation>
</ref>
<ref id="j_infor643_ref_008">
<mixed-citation publication-type="journal"><string-name><surname>Cheng</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Yang</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Yang</surname>, <given-names>Y.</given-names></string-name> (<year>2023</year>). <article-title>Multimodal sentiment analysis based on attentional temporal convolutional network and multi-layer feature fusion</article-title>. <source>IEEE Transactions on Affective Computing</source>, <volume>14</volume>(<issue>4</issue>), <fpage>3149</fpage>–<lpage>3163</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_009">
<mixed-citation publication-type="other"><string-name><surname>Dai</surname>, <given-names>J.</given-names></string-name>, <string-name><given-names>H.</given-names>, <surname>Yan</surname></string-name>, <string-name><given-names>T.</given-names>, <surname>Sun</surname></string-name>, <string-name><given-names>P.</given-names>, <surname>Liu</surname></string-name>, <string-name><given-names>X.</given-names>, <surname>Qiu</surname></string-name> (2021). Does syntax matter? A strong baseline for aspect-based sentiment analysis with RoBERTa. arXiv preprint arXiv:<ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2104.04986">2104.04986</ext-link>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_010">
<mixed-citation publication-type="journal"><string-name><surname>Das</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Singh</surname>, <given-names>T.D.</given-names></string-name> (<year>2023</year>). <article-title>Multimodal sentiment analysis: a survey of methods, trends, and challenges</article-title>. <source>ACM Computing Surveys</source>, <volume>55</volume>(<issue>13s</issue>), <fpage>1</fpage>–<lpage>38</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_011">
<mixed-citation publication-type="book"><string-name><surname>Daubechies</surname>, <given-names>I.</given-names></string-name>, <string-name><surname>Heil</surname>, <given-names>C.</given-names></string-name> (<year>1992</year>). <source>Ten Lectures on Wavelets</source>. <publisher-name>SIAM</publisher-name>, <publisher-loc>Philadelphia</publisher-loc>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_012">
<mixed-citation publication-type="chapter"><string-name><surname>Devlin</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Chang</surname>, <given-names>M.-W.</given-names></string-name>, <string-name><surname>Lee</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Toutanova</surname>, <given-names>K.</given-names></string-name> (<year>2019</year>). <chapter-title>BERT: pre-training of deep bidirectional transformers for language understanding</chapter-title>. In: <source>Proceedings of NAACL-HLT</source>, pp. <fpage>4171</fpage>–<lpage>4186</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_013">
<mixed-citation publication-type="journal"><string-name><surname>Do</surname>, <given-names>H.N.</given-names></string-name>, <string-name><surname>Phan</surname>, <given-names>H.T.</given-names></string-name>, <string-name><surname>Nguyen</surname>, <given-names>N.T.</given-names></string-name> (<year>2024</year>). <article-title>Multimodal sentiment analysis using deep learning and fuzzy logic: a comprehensive survey</article-title>. <source>Applied Soft Computing</source>, <volume>167</volume>, <elocation-id>112279</elocation-id>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_014">
<mixed-citation publication-type="other"><string-name><surname>Dosovitskiy</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Beyer</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>Kolesnikov</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Weissenborn</surname> <given-names>D.</given-names></string-name>, <string-name><surname>Zhai</surname> <given-names>X.</given-names></string-name>, <string-name><surname>Unterthiner</surname> <given-names>T.</given-names></string-name>, <string-name><surname>Dehghani</surname> <given-names>M.</given-names></string-name>, <string-name><surname>Minderer</surname> <given-names>M.</given-names></string-name>, <string-name><surname>Heigold</surname> <given-names>G.</given-names></string-name>, <string-name><surname>Gelly</surname> <given-names>S.</given-names></string-name>, <string-name><surname>Uszkoreit</surname> <given-names>J.</given-names></string-name>, <string-name><surname>Houlsby</surname> <given-names>N.</given-names></string-name> (2020). An image is worth 16×16 words: transformers for image recognition at scale. arXiv preprint arXiv:<ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2010.11929">2010.11929</ext-link>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_015">
<mixed-citation publication-type="journal"><string-name><surname>Feng</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Han</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Shi</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>K.-C.</given-names></string-name> (<year>2026</year>). <article-title>Intrusion detection system for shipping communication networks based on federated distillation learning</article-title>. <source>Expert Systems with Applications</source>, <volume>299</volume>, <fpage>129966</fpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_016">
<mixed-citation publication-type="other"><string-name><surname>Fu</surname>, <given-names>D.Y.</given-names></string-name>, <string-name><surname>Dao</surname>, <given-names>T.</given-names></string-name>, <string-name><surname>Saab</surname>, <given-names>K.K.</given-names></string-name>, <string-name><surname>Thomas</surname>, <given-names>A.W.</given-names></string-name>, <string-name><surname>Rudra</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Ré</surname>, <given-names>C.</given-names></string-name> (2022). Hungry Hungry Hippos: Towards language modeling with state space models. arXiv preprint arXiv:<ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2">2</ext-link>212.14052.</mixed-citation>
</ref>
<ref id="j_infor643_ref_017">
<mixed-citation publication-type="journal"><string-name><surname>Gandhi</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Adhvaryu</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Poria</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Cambria</surname>, <given-names>E.</given-names></string-name>, <string-name><surname>Hussain</surname>, <given-names>A.</given-names></string-name> (<year>2023</year>). <article-title>Multimodal sentiment analysis: a systematic review of history, datasets, multimodal fusion methods, applications, challenges and future directions</article-title>. <source>Information Fusion</source>, <volume>91</volume>, <fpage>424</fpage>–<lpage>444</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_018">
<mixed-citation publication-type="journal"><string-name><surname>Ge</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Yu</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Yue</surname>, <given-names>Q.</given-names></string-name>, <string-name><surname>You</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Zhu</surname>, <given-names>L.</given-names></string-name> (<year>2024</year>). <article-title>MambaTSR: you only need 90k parameters for traffic sign recognition</article-title>. <source>Neurocomputing</source>, <volume>599</volume>, <elocation-id>128104</elocation-id>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_019">
<mixed-citation publication-type="other"><string-name><surname>Gu</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Dao</surname>, <given-names>T.</given-names></string-name> (2023). Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:<ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2312.00752">2312.00752</ext-link>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_020">
<mixed-citation publication-type="other"><string-name><surname>Gu</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Gupta</surname>. <given-names>A.</given-names></string-name>, <string-name><surname>Goel</surname>. <given-names>K.</given-names></string-name>, <string-name><surname>Ré</surname>. <given-names>C.</given-names></string-name> (2022). On the parameterization and initialization of diagonal state space models. Advances in Neural Information Processing Systems 35.</mixed-citation>
</ref>
<ref id="j_infor643_ref_021">
<mixed-citation publication-type="other"><string-name><surname>Gu</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Goel</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Ré</surname>, <given-names>C.</given-names></string-name> (2021). Efficiently modeling long sequences with structured state spaces. arXiv preprint arXiv:<ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2111.00396">2111.00396</ext-link>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_022">
<mixed-citation publication-type="chapter"><string-name><surname>He</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Ren</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Sun</surname>, <given-names>J.</given-names></string-name> (<year>2016</year>). <chapter-title>Deep residual learning for image recognition</chapter-title>. In: <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>, pp. <fpage>770</fpage>–<lpage>778</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_023">
<mixed-citation publication-type="journal"><string-name><surname>Hu</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Yi</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>L.</given-names></string-name> (<year>2024</year>). <article-title>Multichannel cross-modal fusion network for multimodal sentiment analysis considering language information enhancement</article-title>. <source>IEEE Transactions on Industrial Informatics</source>, <volume>20</volume>(<issue>7</issue>), <fpage>9814</fpage>–<lpage>9824</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_024">
<mixed-citation publication-type="journal"><string-name><surname>Huan</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Zhong</surname>, <given-names>G.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>P.</given-names></string-name>, <string-name><surname>Liang</surname>, <given-names>R.</given-names></string-name> (<year>2023</year>). <article-title>UniMF: a unified multimodal framework for multimodal sentiment analysis in missing modalities and unaligned multimodal sequences</article-title>. <source>IEEE Transactions on Multimedia</source>, <volume>26</volume>, <fpage>5753</fpage>–<lpage>5768</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_025">
<mixed-citation publication-type="chapter"><string-name><surname>Huang</surname>, <given-names>S.-C.</given-names></string-name>, <string-name><surname>Shen</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>Lungren</surname>, <given-names>M.P.</given-names></string-name>, <string-name><surname>Yeung</surname>, <given-names>S.</given-names></string-name> (<year>2021</year>). <chapter-title>GLoRIA: a multimodal global-local representation learning framework for label-efficient medical image recognition</chapter-title>. In: <source>Proceedings of the IEEE/CVF International Conference on Computer Vision</source>, pp. <fpage>3922</fpage>–<lpage>3931</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_026">
<mixed-citation publication-type="journal"><string-name><surname>Kang</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Yoo</surname>, <given-names>S.J.</given-names></string-name>, <string-name><surname>Han</surname>, <given-names>D.</given-names></string-name> (<year>2012</year>). <article-title>Senti-lexicon and improved Naïve Bayes algorithms for sentiment analysis of restaurant reviews</article-title>. <source>Expert Systems with Applications</source>, <volume>39</volume>(<issue>5</issue>), <fpage>6000</fpage>–<lpage>6010</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_027">
<mixed-citation publication-type="journal"><string-name><surname>Kim</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Park</surname>, <given-names>S.</given-names></string-name> (<year>2023</year>). <article-title>AOBERT: all-modalities-in-one BERT for multimodal sentiment analysis</article-title>. <source>Information Fusion</source>, <volume>92</volume>, <fpage>37</fpage>–<lpage>45</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_028">
<mixed-citation publication-type="chapter"><string-name><surname>Kim</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Son</surname>, <given-names>B.</given-names></string-name>, <string-name><surname>Kim</surname>, <given-names>I.</given-names></string-name> (<year>2021</year>). <chapter-title>ViLT: vision-and-language transformer without convolution or region supervision</chapter-title>. In: <source>International Conference on Machine Learning</source>. <publisher-name>PMLR</publisher-name>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_029">
<mixed-citation publication-type="chapter"><string-name><surname>Kim</surname>, <given-names>Y.</given-names></string-name> (<year>2014</year>). <chapter-title>Convolutional neural networks for sentence classification</chapter-title>. In: <source>Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing</source>, pp. <fpage>1746</fpage>–<lpage>1751</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_030">
<mixed-citation publication-type="journal"><string-name><surname>Li</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Han</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Shi</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Xin</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>K.-C.</given-names></string-name>, <string-name><surname>Chang</surname>, <given-names>C.-C.</given-names></string-name> (<year>2025</year>a). <article-title>An active client selection scheme based on blockchain for federated learning in shipping</article-title>. <source>IEEE Transactions on Intelligent Transportation Systems</source>, <volume>26</volume>(<issue>11</issue>), <fpage>20669</fpage>–<lpage>20684</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_031">
<mixed-citation publication-type="journal"><string-name><surname>Li</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Han</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Weng</surname>, <given-names>T.-H.</given-names></string-name>, <string-name><surname>Wu</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>K.-C.</given-names></string-name>, <string-name><surname>Castiglione</surname>, <given-names>A.</given-names></string-name> (<year>2025</year>b). <article-title>A secure data storage and sharing scheme for port supply chain based on blockchain and dynamic searchable encryption</article-title>. <source>Computer Standards &amp; Interfaces</source>, <volume>91</volume>, <elocation-id>103887</elocation-id>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_032">
<mixed-citation publication-type="chapter"><string-name><surname>Li</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Liang</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Xie</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Xie</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>K.</given-names></string-name> (<year>2023</year>). <chapter-title>LightNestle: quick and accurate neural sequential tensor completion via meta learning</chapter-title>. In: <source>IEEE INFOCOM 2023-IEEE Conference on Computer Communications</source>. <publisher-name>IEEE</publisher-name>, pp. <fpage>1</fpage>–<lpage>10</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_033">
<mixed-citation publication-type="other"><string-name><surname>Li</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Xu</surname>, <given-names>B.</given-names></string-name>, <string-name><surname>Zhu</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Zhao</surname>, <given-names>T.</given-names></string-name> (2022). CLMLF: a contrastive learning and multi-layer fusion method for multimodal sentiment detection. arXiv preprint arXiv:<ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2204.05515">2204.05515</ext-link>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_034">
<mixed-citation publication-type="journal"><string-name><surname>Liang</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Liang</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Zomaya</surname>, <given-names>A.Y.</given-names></string-name> (<year>2025</year>). <article-title>FedTCTF: tensor completion-based federated learning for device heterogeneity</article-title>. <source>IEEE Transactions on Sustainable Computing</source>, <volume>10</volume>(<issue>6</issue>), <fpage>1227</fpage>–<lpage>1239</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_035">
<mixed-citation publication-type="chapter"><string-name><surname>Lin</surname>, <given-names>T.-Y.</given-names></string-name>, <string-name><surname>Goyal</surname>, <given-names>P.</given-names></string-name>, <string-name><surname>Girshick</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>He</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Dollár</surname>, <given-names>P.</given-names></string-name> (<year>2017</year>). <chapter-title>Focal loss for dense object detection</chapter-title>. In: <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_036">
<mixed-citation publication-type="other"><string-name><surname>Liu</surname>, <given-names>Y.</given-names></string-name> (2019). RoBERTa: a robustly optimized BERT pretraining approach. arXiv preprint arXivurl1907.11692.</mixed-citation>
</ref>
<ref id="j_infor643_ref_037">
<mixed-citation publication-type="journal"><string-name><surname>Lu</surname>, <given-names>Q.</given-names></string-name>, <string-name><surname>Sun</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Long</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Gao</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Feng</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Sun</surname>, <given-names>T.</given-names></string-name> (<year>2024</year>). <article-title>Sentiment analysis: comprehensive reviews, recent advances, and open challenges</article-title>. <source>IEEE Transactions on Neural Networks and Learning Systems</source>, <volume>35</volume>(<issue>11</issue>), <fpage>15092</fpage>–<lpage>15112</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_038">
<mixed-citation publication-type="journal"><string-name><surname>Mai</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Zeng</surname> <given-names>Y.</given-names></string-name>, <string-name><surname>Zheng</surname> <given-names>S.</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>H.</given-names></string-name> (<year>2022</year>). <article-title>Hybrid contrastive learning of tri-modal representation for multimodal sentiment analysis</article-title>. <source>IEEE Transactions on Affective Computing</source>, <volume>14</volume>(<issue>3</issue>), <fpage>2276</fpage>–<lpage>2289</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_039">
<mixed-citation publication-type="journal"><string-name><surname>Mallat</surname>, <given-names>S.G.</given-names></string-name> (<year>1989</year>). <article-title>A theory for multiresolution signal decomposition: the wavelet representation</article-title>. <source>IEEE Transactions on Pattern Analysis and Machine Intelligence</source>, <volume>11</volume>(<issue>7</issue>), <fpage>674</fpage>–<lpage>693</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_040">
<mixed-citation publication-type="journal"><string-name><surname>Moraes</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Valiati</surname>, <given-names>J.F.</given-names></string-name>, <string-name><surname>Neto</surname>, <given-names>W.P.G.</given-names></string-name> (<year>2013</year>). <article-title>Document-level sentiment classification: an empirical comparison between SVM and ANN</article-title>. <source>Expert Systems with Applications</source>, <volume>40</volume>(<issue>2</issue>), <fpage>621</fpage>–<lpage>633</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_041">
<mixed-citation publication-type="chapter"><string-name><surname>Niu</surname>, <given-names>T.</given-names></string-name>, <string-name><surname>Zhu</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Pang</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>El Saddik</surname>, <given-names>A.</given-names></string-name> (<year>2016</year>). <chapter-title>Sentiment analysis on multi-view social data</chapter-title>. In: <source>Proceedings of the 22nd International Conference on Multimedia Modeling</source>. <publisher-name>Springer</publisher-name>, <publisher-loc>Cham</publisher-loc>, pp. <fpage>15</fpage>–<lpage>27</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_042">
<mixed-citation publication-type="journal"><string-name><surname>Pan</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Plaza</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Chanussot</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Hong</surname>, <given-names>D.</given-names></string-name> (<year>2025</year>). <article-title>Hyperspectral image classification with Mamba</article-title>. <source>IEEE Transactions on Geoscience and Remote Sensing</source>, <volume>63</volume>, <elocation-id>5602814</elocation-id>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_043">
<mixed-citation publication-type="journal"><string-name><surname>Pandey</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Vishwakarma</surname>, <given-names>D.K.</given-names></string-name> (<year>2024</year>). <article-title>Progress, achievements, and challenges in multimodal sentiment analysis using deep learning: a survey</article-title>. <source>Applied Soft Computing</source>, <volume>152</volume>, <elocation-id>111206</elocation-id>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_044">
<mixed-citation publication-type="journal"><string-name><surname>Parveen</surname>, <given-names>N.</given-names></string-name>, <string-name><surname>Chakrabarti</surname>, <given-names>P.</given-names></string-name>, <string-name><surname>Hung</surname>, <given-names>B.T.</given-names></string-name>, <string-name><surname>Shaik</surname>, <given-names>A.</given-names></string-name> (<year>2023</year>). <article-title>Twitter sentiment analysis using hybrid gated attention recurrent network</article-title>. <source>Journal of Big Data</source>, <volume>10</volume>(<issue>1</issue>), <fpage>50</fpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_045">
<mixed-citation publication-type="journal"><string-name><surname>Pashchenko</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Rahman</surname>, <given-names>M.F.</given-names></string-name>, <string-name><surname>Hossain</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Uddin</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Islam</surname>, <given-names>T.</given-names></string-name> (<year>2022</year>). <article-title>Emotional and the normative aspects of customers’ reviews</article-title>. <source>Journal of Retailing and Consumer Services</source>, <volume>68</volume>, <fpage>103011</fpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_046">
<mixed-citation publication-type="journal"><string-name><surname>Rodríguez-Ibáñez</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Casánez-Ventura</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Castejón-Mateos</surname>, <given-names>F.</given-names></string-name>, <string-name><surname>Cuenca-Jiménez</surname>, <given-names>P.-M.</given-names></string-name> (<year>2023</year>). <article-title>A review on sentiment analysis from social media platforms</article-title>. <source>Expert Systems with Applications</source>, <volume>223</volume>, <fpage>119862</fpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_047">
<mixed-citation publication-type="journal"><string-name><surname>Shunxiang</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Aoqiang</surname> <given-names>Z.</given-names></string-name>, <string-name><surname>Guangli</surname> <given-names>Z.</given-names></string-name>, <string-name><surname>Zhongliang</surname> <given-names>W.</given-names></string-name>, <string-name><surname>KuanChing</surname> <given-names>L.</given-names></string-name> (<year>2023</year>). <article-title>Building fake review detection model based on sentiment intensity and PU learning</article-title>. <source>IEEE Transactions on Neural Networks and Learning Systems</source>, <volume>34</volume>(<issue>10</issue>), <fpage>6926</fpage>–<lpage>6939</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_048">
<mixed-citation publication-type="journal"><string-name><surname>Singh</surname>, <given-names>U.</given-names></string-name>, <string-name><surname>Abhishek</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Azad</surname>, <given-names>H.K.</given-names></string-name> (<year>2024</year>). <article-title>A survey of cutting-edge multimodal sentiment analysis</article-title>. <source>ACM Computing Surveys</source>, <volume>56</volume>(<issue>9</issue>), <fpage>1</fpage>–<lpage>38</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_049">
<mixed-citation publication-type="journal"><string-name><surname>Su</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Ahmed</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Lu</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Pan</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Bo</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Liu</surname>, <given-names>Y.</given-names></string-name> (<year>2024</year>). <article-title>RoFormer: enhanced transformer with rotary position embedding</article-title>. <source>Neurocomputing</source>, <volume>568</volume>, <fpage>127063</fpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_050">
<mixed-citation publication-type="chapter"><string-name><surname>Sun</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Liu</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Pei</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Zou</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Yan</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Yang</surname>, <given-names>Y.</given-names></string-name> (<year>2024</year>). <chapter-title>CoSeR: bridging image and language for cognitive super-resolution</chapter-title>. In: <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_051">
<mixed-citation publication-type="journal"><string-name><surname>Sun</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Niu</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Yu</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Liu</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>Y.-W.</given-names></string-name> (2025). Multimodal sentiment analysis with mutual Information-based disentangled representation learning. <italic>IEEE Transactions on Affective Computing</italic>, <volume>16</volume>(<issue>3</issue>), <fpage>1606</fpage>–<lpage>1617</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_052">
<mixed-citation publication-type="other"><string-name><surname>Tai</surname>, <given-names>K.S.</given-names></string-name>, <string-name><surname>Socher</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Manning</surname>, <given-names>C.D.</given-names></string-name> (2015). Improved semantic representations from tree-structured long short-term memory networks. arXiv preprint arXiv:<ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1503.00075">1503.00075</ext-link>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_053">
<mixed-citation publication-type="chapter"><string-name><surname>Tang</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Qin</surname>, <given-names>B.</given-names></string-name>, <string-name><surname>Liu</surname>, <given-names>T.</given-names></string-name> (<year>2015</year>). <chapter-title>Document modeling with gated recurrent neural network for sentiment classification</chapter-title>. In: <source>Proceedings of the 2015 Conference on Empirical Methods in Natural Language Processing</source>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_054">
<mixed-citation publication-type="other"><string-name><surname>Tian</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Gao</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Xiao</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Liu</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>He</surname>, <given-names>B.</given-names></string-name>, <string-name><surname>Wu</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Wu</surname>, <given-names>F.</given-names></string-name> (2020). SKEP: sentiment knowledge enhanced pre-training for sentiment analysis. arXiv preprint arXiv:<ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2">2</ext-link>005.05635.</mixed-citation>
</ref>
<ref id="j_infor643_ref_055">
<mixed-citation publication-type="journal"><string-name><surname>Truong</surname>, <given-names>Q.-T.</given-names></string-name>, <string-name><surname>Lauw</surname>, <given-names>H.W.</given-names></string-name> (<year>2019</year>). <article-title>VistaNet: visual aspect attention network for multimodal sentiment analysis</article-title>. <source>Proceedings of the AAAI Conference on Artificial Intelligence</source>, <volume>33</volume>, <fpage>305</fpage>–<lpage>312</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_056">
<mixed-citation publication-type="journal"><string-name><surname>Usama</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Ahmad</surname>, <given-names>B.</given-names></string-name>, <string-name><surname>Song</surname>, <given-names>E.</given-names></string-name>, <string-name><surname>Hossain</surname>, <given-names>M.S.</given-names></string-name>, <string-name><surname>Alrashoud</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Muhammad</surname>, <given-names>G.</given-names></string-name> (<year>2020</year>). <article-title>Attention-based sentiment analysis using convolutional and recurrent neural network</article-title>. <source>Future Generation Computer Systems</source>, <volume>113</volume>, <fpage>571</fpage>–<lpage>578</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_057">
<mixed-citation publication-type="journal"><string-name><surname>Wang</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Liu</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>Q.</given-names></string-name>, <string-name><surname>Tian</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>He</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>Gao</surname>, <given-names>X.</given-names></string-name> (<year>2022</year>). <article-title>Cross-modal enhancement network for multimodal sentiment analysis</article-title>. <source>IEEE Transactions on Multimedia</source>, <volume>25</volume>, <fpage>4909</fpage>–<lpage>4921</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_058">
<mixed-citation publication-type="journal"><string-name><surname>Wang</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Tian</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Liang</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Zhao</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>He</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>Q.</given-names></string-name> (<year>2023</year>a). <article-title>Dual-perspective fusion network for aspect-based multimodal sentiment analysis</article-title>. <source>IEEE Transactions on Multimedia</source>, <volume>26</volume>, <fpage>4028</fpage>–<lpage>4038</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_059">
<mixed-citation publication-type="journal"><string-name><surname>Wang</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Du</surname>, <given-names>Q.</given-names></string-name>, <string-name><surname>Xiang</surname>, <given-names>Y.</given-names></string-name> (<year>2025</year>a). <article-title>Image–text sentiment analysis based on hierarchical interaction fusion and contrast learning enhanced</article-title>. <source>Engineering Applications of Artificial Intelligence</source>, <volume>146</volume>, <elocation-id>110262</elocation-id>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_060">
<mixed-citation publication-type="journal"><string-name><surname>Wang</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Ren</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Yu</surname>, <given-names>Z.</given-names></string-name> (<year>2025</year>b). <article-title>Multimodal sentiment analysis based on multiple attention</article-title>. <source>Engineering Applications of Artificial Intelligence</source>, <volume>140</volume>, <elocation-id>109731</elocation-id>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_061">
<mixed-citation publication-type="journal"><string-name><surname>Wang</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Yang</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Jiang</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Ma</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Xie</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>T.</given-names></string-name> (<year>2024</year>). <article-title>Cross-modal incongruity aligning and collaborating for multi-modal sarcasm detection</article-title>. <source>Information Fusion</source>, <volume>103</volume>, <elocation-id>102132</elocation-id>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_062">
<mixed-citation publication-type="journal"><string-name><surname>Wang</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>Niu</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Yu</surname>, <given-names>S.</given-names></string-name> (<year>2019</year>). <article-title>SentiDiff: combining textual information and sentiment diffusion patterns for Twitter sentiment analysis</article-title>. <source>IEEE Transactions on Knowledge and Data Engineering</source>, <volume>32</volume>(<issue>10</issue>), <fpage>2026</fpage>–<lpage>2039</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_063">
<mixed-citation publication-type="journal"><string-name><surname>Wang</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Cui</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>Y.</given-names></string-name> (<year>2025</year>c). <article-title>Manifold knowledge-guided feature fusion network for multimodal sentiment analysis</article-title>. <source>Expert Systems with Applications</source>, <volume>280</volume>, <elocation-id>127537</elocation-id>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_064">
<mixed-citation publication-type="journal"><string-name><surname>Wang</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Huang</surname>, <given-names>G.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>H.</given-names></string-name> (<year>2023</year>b). <article-title>Automatically constructing a fine-grained sentiment lexicon for sentiment analysis</article-title>. <source>Cognitive Computation</source>, <volume>15</volume>(<issue>1</issue>), <fpage>254</fpage>–<lpage>271</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_065">
<mixed-citation publication-type="journal"><string-name><surname>Wang</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Jian</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Zhuang</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Guo</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Leng</surname>, <given-names>Y.</given-names></string-name> (<year>2025</year>d). <article-title>SSLMM: semi-supervised learning with missing modalities for multimodal sentiment analysis</article-title>. <source>Information Fusion</source>, <volume>120</volume>, <fpage>103058</fpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_066">
<mixed-citation publication-type="chapter"><string-name><surname>Wei</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Yuan</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Yang</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Shen</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>M.</given-names></string-name> (<year>2023</year>). <chapter-title>Tackling modality heterogeneity with multi-view calibration network for multimodal sentiment detection</chapter-title>. In: <source>Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics</source>, pp. <fpage>5240</fpage>–<lpage>5252</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_067">
<mixed-citation publication-type="journal"><string-name><surname>Wu</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Kong,</surname> <given-names>D.</given-names></string-name>, <string-name><surname>Wang,</surname> <given-names>L.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Han</surname>, <given-names>Y.</given-names></string-name> (<year>2025</year>). <article-title>Multimodal sentiment analysis method based on image-text quantum transformer</article-title>. <source>Neurocomputing</source>, <volume>637</volume>, <fpage>130107</fpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_068">
<mixed-citation publication-type="journal"><string-name><surname>Xie</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Cui</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Tan</surname>, <given-names>T.</given-names></string-name>, <string-name><surname>Zheng</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Yu</surname>, <given-names>Z.</given-names></string-name> (<year>2024</year>). <article-title>FusionMamba: dynamic feature enhancement for multimodal image fusion with Mamba</article-title>. <source>Visual Intelligence</source>, <volume>2</volume>(<issue>1</issue>), <fpage>37</fpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_069">
<mixed-citation publication-type="chapter"><string-name><surname>Xu</surname>, <given-names>N.</given-names></string-name> (<year>2017</year>). <chapter-title>Analyzing multimodal public sentiment based on hierarchical semantic attentional network</chapter-title>. In: <source>Proceedings of the 2017 IEEE International Conference on Intelligence and Security Informatics (ISI’17)</source>, pp. <fpage>152</fpage>–<lpage>154</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_070">
<mixed-citation publication-type="chapter"><string-name><surname>Xu</surname>, <given-names>N.</given-names></string-name>, <string-name><surname>Mao</surname>, <given-names>W.</given-names></string-name> (<year>2017</year>). <chapter-title>MultiSentiNet: a deep semantic network for multimodal sentiment analysis</chapter-title>. In: <source>Proceedings of the 2017 ACM on Conference on Information and Knowledge Management</source>, pp. <fpage>2399</fpage>–<lpage>2402</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_071">
<mixed-citation publication-type="chapter"><string-name><surname>Xu</surname>, <given-names>N.</given-names></string-name>, <string-name><surname>Mao</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>G.</given-names></string-name> (<year>2018</year>). <chapter-title>A co-memory network for multimodal sentiment analysis</chapter-title>. In: <source>Proceedings of the 41st International ACM SIGIR Conference on Research and Development in Information Retrieval</source>, pp. <fpage>929</fpage>–<lpage>932</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_072">
<mixed-citation publication-type="journal"><string-name><surname>Xue</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Niu</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Wu</surname>, <given-names>X.</given-names></string-name> (<year>2022</year>). <article-title>Multi-level attention map network for multimodal sentiment analysis</article-title>. <source>IEEE Transactions on Knowledge and Data Engineering</source>, <volume>35</volume>(<issue>5</issue>), <fpage>5105</fpage>–<lpage>5118</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_073">
<mixed-citation publication-type="chapter"><string-name><surname>Yang</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Feng</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>D.</given-names></string-name> (<year>2021</year>). <chapter-title>Multimodal sentiment detection based on multi-channel graph neural networks</chapter-title>. In: <source>Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing</source>, pp. <fpage>328</fpage>–<lpage>339</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_074">
<mixed-citation publication-type="chapter"><string-name><surname>Ye</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Shan</surname>, <given-names>H.</given-names></string-name> (<year>2025</year>). <chapter-title>DepMamba: progressive fusion mamba for multimodal depression detection</chapter-title>. In: <source>ICASSP 2025–2025 IEEE International Conference on Acoustics, Speech and Signal Processing</source>. <publisher-name>IEEE</publisher-name>, pp. <fpage>1</fpage>–<lpage>5</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_075">
<mixed-citation publication-type="journal"><string-name><surname>Zhai</surname>, <given-names>G.</given-names></string-name>, <string-name><surname>Yang</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Du</surname>, <given-names>S.</given-names></string-name> (<year>2020</year>). <article-title>Multi-attention fusion for sentiment analysis of educational big data</article-title>. <source>Big Data Mining and Analytics</source>, <volume>3</volume>(<issue>4</issue>), <fpage>311</fpage>–<lpage>319</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_076">
<mixed-citation publication-type="journal"><string-name><surname>Zhang</surname>, <given-names>B.</given-names></string-name>, <string-name><surname>Yuan</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Xu</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Gao</surname>, <given-names>K.</given-names></string-name> (2024). <article-title>Crossmodal translation based meta weight adaption for robust image-text sentiment analysis</article-title>. <source>IEEE Transactions on Multimedia</source>, <volume>26</volume>, <fpage>9949</fpage>–<lpage>9961</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_077">
<mixed-citation publication-type="journal"><string-name><surname>Zhang</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Liu</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Jiao</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>K.</given-names></string-name> (<year>2025</year>a). <article-title>A multimodal semantic fusion network with cross-modal alignment for multimodal sentiment analysis</article-title>. <source>ACM Transactions on Multimedia Computing, Communications, and Applications</source>, <volume>21</volume>(<issue>10</issue>), <fpage>1</fpage>–<lpage>22</lpage>. <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1145/3744648" xlink:type="simple">https://doi.org/10.1145/3744648</ext-link>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_078">
<mixed-citation publication-type="journal"><string-name><surname>Zhang</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Duan</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Wei</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>K.-C.</given-names></string-name> (<year>2025</year>b). <article-title>Textual graph representation with syntactic weighting for implicit sentiment analysis</article-title>. <source>IEEE Transactions on Emerging Topics in Computational Intelligence</source>, <volume>9</volume>(<issue>6</issue>), <fpage>4288</fpage>–<lpage>4299</lpage>. <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/TETCI.2025.3550515" xlink:type="simple">https://doi.org/10.1109/TETCI.2025.3550515</ext-link>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_079">
<mixed-citation publication-type="chapter"><string-name><surname>Zhang</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>H.</given-names></string-name> (<year>2023</year>). <chapter-title>Cross-modal sentiment analysis based on transformer and image-text collaborative interaction</chapter-title>. In: <source>2023 International Conference on Computer Engineering and Distance Learning (CEDL)</source>. <publisher-name>IEEE</publisher-name>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_080">
<mixed-citation publication-type="journal"><string-name><surname>Zhao</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Yao</surname> <given-names>X.</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>J.</given-names></string-name>, <string-name><surname>Jia</surname> <given-names>G.</given-names></string-name>, <string-name><surname>Ding</surname> <given-names>G.</given-names></string-name>, <string-name><surname>Chua</surname> <given-names>T.-S.</given-names></string-name>, <string-name><surname>Schuller</surname> <given-names>B.W.</given-names></string-name>, <string-name><surname>Keutzer</surname> <given-names>K.</given-names></string-name> (<year>2021</year>). <article-title>Affective image content analysis: two decades review and new perspectives</article-title>. <source>IEEE Transactions on Pattern Analysis and Machine Intelligence</source>, <volume>44</volume>(<issue>10</issue>), <fpage>6729</fpage>–<lpage>6757</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_081">
<mixed-citation publication-type="journal"><string-name><surname>Zhao</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Poria</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Tang</surname>, <given-names>B.</given-names></string-name> (<year>2025</year>a). <article-title>Toward robust multimodal sentiment analysis using multimodal foundational models</article-title>. <source>Expert Systems with Applications</source>, <volume>276</volume>, <fpage>126974</fpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_082">
<mixed-citation publication-type="journal"><string-name><surname>Zhao</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>K.-C.</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Ye</surname>, <given-names>T.</given-names></string-name> (<year>2025</year>b). <article-title>Hyperbolic graph attention network fusing long-context for technical keyphrase extraction</article-title>. <source>Information Fusion</source>, <volume>120</volume>, <fpage>103061</fpage>. <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1016/j.inffus.2025.103061" xlink:type="simple">https://doi.org/10.1016/j.inffus.2025.103061</ext-link>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_083">
<mixed-citation publication-type="journal"><string-name><surname>Zhao</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Zhu</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Xue</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Liu</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Tian</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Heng Chua</surname>, <given-names>M.C.</given-names></string-name>, <string-name><surname>Liu</surname>, <given-names>M.</given-names></string-name> (<year>2019</year>). <article-title>An image-text consistency driven multimodal sentiment analysis approach for social media</article-title>. <source>Information Processing &amp; Management</source>, <volume>56</volume>(<issue>6</issue>), <fpage>102097</fpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_084">
<mixed-citation publication-type="chapter"><string-name><surname>Zhou</surname>, <given-names>P.</given-names></string-name>, <string-name><surname>Shi</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Tian</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Qi</surname> <given-names>Z.</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B.</given-names></string-name>, <string-name><surname>Hao</surname> <given-names>H.</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>B.</given-names></string-name> (<year>2016</year>). <chapter-title>Attention-based bidirectional long short-term memory networks for relation classification</chapter-title>. In: <source>Proceedings of the 54th Annual Meeting of the Association for Computational Linguistics</source>, pp. <fpage>207</fpage>–<lpage>212</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_085">
<mixed-citation publication-type="other"><string-name><surname>Zhou</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Liang</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Zomaya</surname>, <given-names>A.Y.</given-names></string-name> (2026a). MultiSecDFL: multi-metric filtering-based secure aggregation method for decentralized federated learning. <italic>IEEE Transactions on Consumer Electronics</italic>. <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/TCE.2026.3688316" xlink:type="simple">https://doi.org/10.1109/TCE.2026.3688316</ext-link>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_086">
<mixed-citation publication-type="journal"><string-name><surname>Zhou</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Liang</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Meng</surname>, <given-names>W.</given-names></string-name> (<year>2026</year>b). <article-title>TrustHFL: an efficient aggregation method for trustworthy hierarchical federated learning</article-title>. <source>IEEE Internet of Things Journal</source>, <volume>13</volume>(<issue>11</issue>), <fpage>24618</fpage>–<lpage>24631</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_087">
<mixed-citation publication-type="journal"><string-name><surname>Zhou</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Chen</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Yang</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Liang</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Zomaya</surname>, <given-names>A.Y.</given-names></string-name> (<year>2024</year>). <article-title>TrustBCFL: mitigating data bias in IoT through blockchain-enabled federated learning</article-title>. <source>IEEE Internet of Things Journal</source>, <volume>11</volume>(<issue>15</issue>), <fpage>25648</fpage>–<lpage>25662</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_088">
<mixed-citation publication-type="journal"><string-name><surname>Zhu</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>Zhu</surname>, <given-names>Z.</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Xu</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Kong</surname>, <given-names>X.</given-names></string-name> (<year>2023</year>). <article-title>Multimodal sentiment analysis based on fusion methods: a survey</article-title>. <source>Information Fusion</source>, <volume>95</volume>, <fpage>306</fpage>–<lpage>325</lpage>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_089">
<mixed-citation publication-type="chapter"><string-name><surname>Zhu</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>Liao</surname>, <given-names>B.</given-names></string-name>, <string-name><surname>Zhang</surname>, <given-names>Q.</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Liu</surname>, <given-names>W.</given-names></string-name>, <string-name><surname>Wang</surname>, <given-names>X.</given-names></string-name> (<year>2024</year>). <chapter-title>Vision Mamba: efficient visual representation learning with bidirectional state space model</chapter-title>. In: <source>International Conference on Machine Learning</source>.</mixed-citation>
</ref>
<ref id="j_infor643_ref_090">
<mixed-citation publication-type="journal"><string-name><surname>Zhu</surname>, <given-names>T.</given-names></string-name>, <string-name><surname>Li</surname>, <given-names>L.</given-names></string-name>, <string-name><surname>Yang</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Zhao</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Liu</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Qian</surname>, <given-names>J.</given-names></string-name> (<year>2022</year>). <article-title>Multimodal sentiment analysis with image-text interaction network</article-title>. <source>IEEE Transactions on Multimedia</source>, <volume>25</volume>, <fpage>3375</fpage>–<lpage>3385</lpage>.</mixed-citation>
</ref>
</ref-list>
</back>
</article>
