This repository hosts the
SE-DiCoW model developed by
BUT Speech@FIT, in collaboration with
JHU CLSP/HLTCOE and
CMU LTI, tailored for
target-speaker multi-talker automatic speech recognition (TS-ASR).
1from transformers import AutoModelForSpeechSeq2Seq
2
3MODEL_NAME = "BUT-FIT/SE_DiCoW"
4model = AutoModelForSpeechSeq2Seq.from_pretrained(MODEL_NAME, trust_remote_code=True)
-
📰 ICASSP 2026:
SE-DiCoW: Self-Enrolled Diarization-Conditioned Whisper
[IEEE ICASSP 2026]
-
📰
Journal Paper (CSL 2026):
DiCoW: Diarization-Conditioned Whisper for Target Speaker ASR
Computer Speech & Language, 2026
-
📰
ICASSP 2025:
Target Speaker ASR with Whisper
IEEE ICASSP 2025
1@INPROCEEDINGS{polok2026sedicow,
2 author={Polok, Alexander and Klement, Dominik and Cornell, Samuele and Wiesner, Matthew and Černocký, Jan and Khudanpur, Sanjeev and Burget, Lukáš},
3 booktitle={ICASSP 2026 - 2026 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)},
4 title={SE-DiCoW: Self-Enrolled Diarization-Conditioned Whisper},
5 year={2026},
6 pages={1-5},
7}
8
9@article{POLOK2026101841,
10 title = {DiCoW: Diarization-conditioned Whisper for target speaker automatic speech recognition},
11 journal = {Computer Speech & Language},
12 volume = {95},
13 pages = {101841},
14 year = {2026},
15 doi = {https://doi.org/10.1016/j.csl.2025.101841},
16 author = {Alexander Polok and Dominik Klement and Martin Kocour and Jiangyu Han and Federico Landini and Bolaji Yusuf and Matthew Wiesner and Sanjeev Khudanpur and Jan Černocký and Lukáš Burget},
17}
18
19@INPROCEEDINGS{10887683,
20 author={Polok, Alexander and Klement, Dominik and Wiesner, Matthew and Khudanpur, Sanjeev and Černocký, Jan and Burget, Lukáš},
21 booktitle={ICASSP 2025},
22 title={Target Speaker ASR with Whisper},
23 year={2025},
24 doi={10.1109/ICASSP49660.2025.10887683}
25}
🏢
Affiliation: BUT Speech@FIT, Brno University of Technology