@inproceedings{49443666f0604b9fbe25d4c4911777e4,
title = "SV-Mixer: Replacing the Transformer Encoder with Lightweight MLPs for Self-Supervised Model Compresison in Speaker Verification",
abstract = "Self-supervised learning (SSL) has pushed speaker verification accuracy close to state-of-the-art levels, but the Transformer backbones used in most SSL encoders hinder on-device and real-time deployment. Prior compression work trims layer depth or width yet still inherits the quadratic cost of self-attention. We propose SV-Mixer, the first fully MLPbased student encoder for SSL distillation. SV-Mixer replaces Transformer with three lightweight modules: Multi-Scale Mixing for multi-resolution temporal features, Local-Global Mixing for frame-to-utterance context, and Group Channel Mixing for spectral subspaces. Distilled from WavLM, SV-Mixer outperforms a Transformer student by 14.6 \% while cutting parameters and GMACs by over half, and at 7 5 \% compression, it closely matches the teacher's performance. Our results show that attention-free SSL students can deliver teacher-level accuracy with hardwarefriendly footprints, opening the door to robust on-device speaker verification.",
keywords = "knowledge distillation, mlp-mixer, model compression, speaker verification, transformer-free architecture",
author = "Jungwoo Heo and Shin, \{Hyun Seo\} and Lim, \{Chan Yeong\} and Koo, \{Kyo Won\} and Kim, \{Seung Bin\} and Jisoo Son and Yu, \{Ha Jin\}",
note = "Publisher Copyright: {\textcopyright} 2025 IEEE.; 2025 IEEE Automatic Speech Recognition and Understanding Workshop, ASRU 2025 ; Conference date: 06-12-2025 Through 10-12-2025",
year = "2025",
doi = "10.1109/ASRU65441.2025.11434806",
language = "English",
series = "ASRU 2025 - 2025 IEEE Automatic Speech Recognition and Understanding Workshop",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
booktitle = "ASRU 2025 - 2025 IEEE Automatic Speech Recognition and Understanding Workshop",
address = "United States",
}