<?xml version="1.0" encoding="UTF-8"?>

<?xml-stylesheet type="text/xsl" href="/static/oaitohtml.xsl"?>

<!--
<?xml-stylesheet type="text/xsl" href="/oaitohtml.xsl"?>
-->

<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
    <responseDate>2026-10-12T04:44:47Z</responseDate>
    <request verb="GetRecord" metadataPrefix="oai_dc" identifier="10.57760/sciencedb.27127" >https://www.scidb.cn/oai</request>
<GetRecord>
    <record>
    <header >
    <identifier>10.57760/sciencedb.27127</identifier>
    <datestamp>2025-09-04T12:35:07Z</datestamp>
</header>
    <metadata>
        
<oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:date>2025-09-04</dc:date>
  <dc:title>Dual-Teacher Multimodal Fusion for Depression&amp;nbsp;Detection</dc:title>
  <dc:identifier>doi:10.57760/sciencedb.27127</dc:identifier>
  <dc:language>en</dc:language>
  <dc:description>Depression affects over 264 million people worldwide and is a leading cause of disability. Existing machine learning approaches for analyzing behavioral data&amp;mdash;including electronic health records, wearable devices, and social media&amp;mdash;remain constrained by unimodal paradigms. Language models often ignore paralinguistic biomarkers such as speech prosody, while audio-based methods overlook semantic context, leaving critical cross-modal interactions unexploited. Here, we introduce a Teacher-Student Architecture-based Multimodal Fusion Network (TSA-MFN) that fundamentally advances depression classification. Dual specialized teacher models extract high-confidence text semantics and audio biomarkers, which are integrated by a student network through multi-head attention mechanisms. Fusion weights are dynamically computed via learnable similarity matrices, and a hybrid loss function balances knowledge transfer with classification optimization, enabling phased teacher-student synergy during training. TSA-MFN achieves a 99.1% F1-score on the DAIC-WOZ dataset, surpassing unimodal baselines by 18.7% and conventional fusion models by 12.4%, with ablation studies confirming the contribution of each component. By combining dual-teacher guidance with adaptive similarity-based fusion, this framework establishes a new paradigm for interpretable and balanced multimodal learning in mental health analytics.</dc:description>
  <dc:subject>Multimodal fusion; LLM; Teacher-student architecture; Depression detection; Machine learning based health-care</dc:subject>
  <dc:creator>Lin Gan</dc:creator>
  <dc:rights>PUBLIC</dc:rights>
  <dc:rights>https://api.github.com/licenses/cc0-1.0</dc:rights>
  <dc:type>dataset</dc:type>
  <dc:publisher>Science Data Bank</dc:publisher>
</oai_dc:dc>

    </metadata>
</record>
</GetRecord>
</OAI-PMH>