@inproceedings{21d1a0031f4e40479f3e20e425b4cbcc,
title = "Token-based Attractors and Cross-attention in Spoof Diarization",
abstract = "Spoof diarization identifies 'what spoofed when' in a given speech by temporally locating spoofed regions and determining their manipulation techniques. As a first step toward this task, prior work proposed a two-branch model for localization and spoof type clustering, which laid the foundation for spoof diarization. However, its simple structure limits the ability to capture complex spoofing patterns and lacks explicit reference points for distinguishing between bona fide and various spoofing types. To address these limitations, our approach introduces learnable tokens where each token represents acoustic features of bona fide and spoofed speech. These attractors interact with frame-level embeddings to extract discriminative representations, improving separation between genuine and generated speech. Vast experiments on PartialSpoof dataset consistently demon-strate that our approach outperforms existing methods in bona fide detection and spoofing method clustering.",
keywords = "Attractor tokens, Cross-attention, Speaker diarization, Spoof diarization",
author = "Koo, \{Kyo Won\} and Lim, \{Chan Yeong\} and Jun, \{Jee Weon\} and Shim, \{Hye Jin\} and Yu, \{Ha Jin\}",
note = "Publisher Copyright: {\textcopyright} 2025 IEEE.; 2025 IEEE Automatic Speech Recognition and Understanding Workshop, ASRU 2025 ; Conference date: 06-12-2025 Through 10-12-2025",
year = "2025",
doi = "10.1109/ASRU65441.2025.11434759",
language = "English",
series = "ASRU 2025 - 2025 IEEE Automatic Speech Recognition and Understanding Workshop",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
booktitle = "ASRU 2025 - 2025 IEEE Automatic Speech Recognition and Understanding Workshop",
address = "United States",
}