PageSourceSearch

https://www.isca-archive.org/interspeech_2024/index.html

html isca-archive.org collected 2026-09-24 08:39:34 UTC 1,019,016 bytes, 11,173 lines download raw bytes

1<!DOCTYPE html>
2<html>
3    <head>
4        <meta charset="UTF-8">
5        <title>ISCA Archive</title>
6        <meta name="viewport" content="width=device-width, initial-scale=1">
7
8
9        <!-- JQuery helpers -->
10        <link rel="stylesheet" href="../resources/jquery.dataTables.min.css">
11        
11<script src="../resources/jquery-3.5.1.min.js"></script>
11
12        
12<script src="../resources/jquery.dataTables.min.js"></script>
12
13        
13<script src="../resources/accent-neutralise.js"></script>
13
14
15        <!-- Fonts -->
16        <link rel="stylesheet" href="../resources/fontawesome-free-subset/style.css">
17
18        <!-- Overall theme -->
19        <link rel="stylesheet" href="../resources/w3.css">
20        <link rel="stylesheet" href="../resources/w3-theme-blue.css">
21        
21<script src="../resources/w3.js"></script>
21
22
23        <!-- ISCA archive publication requirements -->
24        <base href="https://www.isca-archive.org/interspeech_2024/" />
25        <link rel="stylesheet" href="../resources/is.css">
26    </head>
27    <body>
28
29        <!-- Top menu bar, fixed on large/medium screens -->
30        <div class="w3-top w3-hide-small">
31            <div class="w3-bar w3-grayscale-min w3-theme-d4 w3-center">
32                <a href="../../index.html" class="w3-bar-item w3-button w3-theme-d2 w3-mobile">
33                    <i class="icon-home w3-margin-right"></i>ISCA
34                </a>
35                <a href="../index.html" class="w3-bar-item w3-button w3-mobile">Archive</a>
36                <a href="#" class="w3-bar-item w3-button w3-mobile">Interspeech 2024</a>
37                <a class="w3-bar-item w3-button w3-mobile" onclick="document.getElementById('sessionchooser').style.display='block'">Sessions
38                </a>
39                <a href="#bypaper" class="w3-bar-item w3-button w3-mobile"><i class="icon-search" style='margin-right:5px'></i>Search</a>
40                <a href="https://interspeech2024.org/" class="w3-bar-item w3-button w3-mobile w3-right">Website</a>
41                <a href="booklet.pdf" class="w3-bar-item w3-button w3-mobile w3-right">Booklet</a>
42            </div>
43        </div>
44
45        <!-- Top menu bar, scrollable on small screens -->
46        <div class="w3-hide-large w3-hide-medium">
47            <div class="w3-bar w3-grayscale-min w3-theme-d4 w3-center">
48                <span class="w3-bar-item w3-button w3-mobile w3-opacity-max">&nbsp;</span>
49                <a href="../../index.html" class="w3-bar-item w3-button w3-mobile">
50                    <i class="icon-home w3-margin-right"></i>ISCA
51                </a>
52                <a href="../index.html" class="w3-bar-item w3-button w3-mobile">Archive</a>
53                <a href="#" class="w3-bar-item w3-button w3 w3-mobile" onclick="document.getElementById('sessionchooser').style.display='block'">Sessions
54                </a>
55                <a href="#bypaper" class="w3-bar-item w3-button w3-mobile"><i class="icon-search" style='margin-right:5px'></i>Search</a>
56                <a href="https://interspeech2024.org/" class="w3-bar-item w3-button w3-mobile w3-right">Website</a>
57                <a href="booklet.pdf" class="w3-bar-item w3-button w3-mobile w3-right">Booklet</a>
58            </div>
59        </div>
60
61        <!-- Papers help popup -->
62        <div id="help_papers" class="w3-modal">
63            <div class="w3-modal-content w3-card-4 w3-greyscale w3-theme-d4 w3-padding w3-bordered">
64                <div class="w3-container">
65                    <span onclick="document.getElementById('help_papers').style.display='none'"
66                          class="w3-button w3-display-topright">&times;</span>
67                    <div class="w3-container">
68                        <p class="w3-text">Click on column names to sort.</p>
69                        <p class="w3-text">Searching uses the 'and' of terms e.g. <span class='w3-monospace'>Smith Interspeech</span> matches all papers by Smith in any Interspeech. The order of terms is not significant.</p>
70                        <p class="w3-text">Use double quotes for exact phrasal matches e.g. <span class='w3-monospace'>"acoustic features"</span>.</p>
71                        <p class="w3-text">Case is ignored.</p>
72                        <p class="w3-text">Diacritics are optional e.g. <span class='w3-monospace'>lefevre</span> also matches <span class='w3-monospace'>lefèvre</span> (but not vice versa).</p>
73                        <p class="w3-text">It can be useful to turn off spell-checking for the search box in your browser preferences.</p>
74                        <p class="w3-text">If you prefer to scroll rather than page, increase the number in the show entries dropdown.</p>
75                    </div>
76                </div>
77            </div>
78        </div>
79
80
81        <div class="w3-top w3-hide-medium w3-hide-large">
82            <div class="w3-bar w3-grayscale-min w3-theme-d4 w3-opacity-max">
83                <a href="#" class="w3-bar-item w3-button w3-theme-d2 w3-left">top</a>
84            </div>
85        </div>
86
87        <div class="w3-grayscale w3-theme-l5">
88
89            <!-- Conference header -->
90            <div class="w3-container" id="about">
91                <div class="w3-content" style="max-width:1100px;margin-top:50px; margin-bottom: 10px">
92                    <h2 class="w3-center w3-padding-16">
93                        <span class="w3-text">Interspeech 2024</span>
94                    </h2>
95                    <h5 class="w3-text w3-center"> Kos, Greece<br>1-5 September 2024</h5>
96                    <br>
97                    <h5 class="w3-text w3-center">General Chairs: Itshak Lapidot, Sharon Gannot</h5>
98                    <br>
99                    <h5 class="w3-text w3-center">Technical Program Chairs: Jean-François Bonastre, Luciana Ferrer, Reinhold Haeb-Umbach</h5>
100                    <pre class="w3-text w3-center">doi: 10.21437/Interspeech.2024</pre>
101                    <pre class="w3-text w3-center">ISSN: 2958-1796</pre>
102                </div>
103            </div>
104
105            <!-- Sessions -->
106            <div class="w3-container">
107                <div class="w3-content" style="max-width:1200px;margin-top: 10px">
108                    <div class="w3-content" style="height:10px"  id="Keynote 1 ISCA Medallist"></div>
109                    <div class="w3-card w3-round w3-white w3-padding">
110                        <div class="w3-container"  style="margin-top:40px">
111                            <h4 class="w3-center">Keynote 1 ISCA Medallist</h4>
112                            <hr>
113                            <a class="w3-text" href="trancoso24_interspeech.html">
114                                <p>
115                                    Towards Responsible Speech Processing
116                                    <br>
117                                    <span class="w3-text w3-text-theme">
118                                        Isabel Trancoso
119                                    </span>
120                                </p>
121                            </a>
122                        </div>
123                    </div>
124                    <br>
125                    <div class="w3-content" style="height:10px"  id="L2 Speech, Bilingualism and Code-Switching"></div>
126                    <div class="w3-card w3-round w3-white w3-padding">
127                        <div class="w3-container"  style="margin-top:40px">
128                            <h4 class="w3-center">L2 Speech, Bilingualism and Code-Switching</h4>
129                            <hr>
130                            <a class="w3-text" href="wesolek24_interspeech.html">
131                                <p>
132                                    The influence of L2 accent strength and different error types on personality trait ratings
133                                    <br>
134                                    <span class="w3-text w3-text-theme">
135                                        Sarah Wesolek, Piotr Gulgowski, Joanna Blaszczak, Marzena Zygis
136                                    </span>
137                                </p>
138                            </a>
139                            <a class="w3-text" href="chi24_interspeech.html">
140                                <p>
141                                    Characterizing code-switching: Applying Linguistic Principles for Metric Assessment and Development
142                                    <br>
143                                    <span class="w3-text w3-text-theme">
144                                        Jie Chi, Electra Wallington, Peter Bell
145                                    </span>
146                                </p>
147                            </a>
148                            <a class="w3-text" href="xue24_interspeech.html">
149                                <p>
150                                    Towards a better understanding of receptive multilingualism: listening conditions and priming effects
151                                    <br>
152                                    <span class="w3-text w3-text-theme">
153                                        Wei Xue, Ivan Yuen, Bernd Möbius
154                                    </span>
155                                </p>
156                            </a>
157                            <a class="w3-text" href="mohapatra24b_interspeech.html">
158                                <p>
159                                    2.5D Vocal Tract Modeling: Bridging Low-Dimensional Efficiency with 3D Accuracy
160                                    <br>
161                                    <span class="w3-text w3-text-theme">
162                                        Debasish Ray Mohapatra, Victor Zappi, Sidney Fels
163                                    </span>
164                                </p>
165                            </a>
166                        </div>
167                    </div>
168                    <br>
169                    <div class="w3-content" style="height:10px"  id="Speaker Diarization 1"></div>
170                    <div class="w3-card w3-round w3-white w3-padding">
171                        <div class="w3-container"  style="margin-top:40px">
172                            <h4 class="w3-center">Speaker Diarization 1</h4>
173                            <hr>
174                            <a class="w3-text" href="chowdhury24_interspeech.html">
175                                <p>
176                                    Investigating Confidence Estimation Measures for Speaker Diarization
177                                    <br>
178                                    <span class="w3-text w3-text-theme">
179                                        Anurag Chowdhury, Abhinav Misra, Mark C. Fuhs, Monika Woszczyna
180                                    </span>
181                                </p>
182                            </a>
183                            <a class="w3-text" href="li24x_interspeech.html">
184                                <p>
185                                    Speakers Unembedded: Embedding-free Approach to Long-form Neural Diarization
186                                    <br>
187                                    <span class="w3-text w3-text-theme">
188                                        Xiang Li, Vivek Govindan, Rohit Paturi, Sundararajan Srinivasan
189                                    </span>
190                                </p>
191                            </a>
192                            <a class="w3-text" href="huang24d_interspeech.html">
193                                <p>
194                                    On the Success and Limitations of Auxiliary Network Based Word-Level End-to-End Neural Speaker Diarization
195                                    <br>
196                                    <span class="w3-text w3-text-theme">
197                                        Yiling Huang, Weiran Wang, Guanlong Zhao, Hank Liao, Wei Xia, Quan Wang
198                                    </span>
199                                </p>
200                            </a>
201                            <a class="w3-text" href="harkonen24_interspeech.html">
202                                <p>
203                                    EEND-M2F: Masked-attention mask transformers for speaker diarization
204                                    <br>
205                                    <span class="w3-text w3-text-theme">
206                                        Marc Härkönen, Samuel J. Broughton, Lahiru Samarakoon
207                                    </span>
208                                </p>
209                            </a>
210                            <a class="w3-text" href="yin24_interspeech.html">
211                                <p>
212                                    AFL-Net: Integrating Audio, Facial, and Lip Modalities with a Two-step Cross-attention for Robust Speaker Diarization in the Wild
213                                    <br>
214                                    <span class="w3-text w3-text-theme">
215                                        YongKang Yin, Xu Li, Ying Shan, YueXian Zou
216                                    </span>
217                                </p>
218                            </a>
219                            <a class="w3-text" href="arya24_interspeech.html">
220                                <p>
221                                    Exploiting Wavelet Scattering Transform for an Unsupervised Speaker Diarization in Deep Neural Network Framework
222                                    <br>
223                                    <span class="w3-text w3-text-theme">
224                                        Arunav Arya, Murtiza Ali, Karan Nathwani
225                                    </span>
226                                </p>
227                            </a>
228                        </div>
229                    </div>
230                    <br>
231                    <div class="w3-content" style="height:10px"  id="Speech and Audio Analysis and Representations"></div>
232                    <div class="w3-card w3-round w3-white w3-padding">
233                        <div class="w3-container"  style="margin-top:40px">
234                            <h4 class="w3-center">Speech and Audio Analysis and Representations</h4>
235                            <hr>
236                            <a class="w3-text" href="zhao24h_interspeech.html">
237                                <p>
238                                    MINT: Boosting Audio-Language Model via Multi-Target Pre-Training and Instruction Tuning
239                                    <br>
240                                    <span class="w3-text w3-text-theme">
241                                        Hang Zhao, Yifei Xin, Zhesong Yu, Bilei Zhu, Lu Lu, Zejun Ma
242                                    </span>
243                                </p>
244                            </a>
245                            <a class="w3-text" href="niizumi24_interspeech.html">
246                                <p>
247                                    M2D-CLAP: Masked Modeling Duo Meets CLAP for Learning General-purpose Audio-Language Representation
248                                    <br>
249                                    <span class="w3-text w3-text-theme">
250                                        Daisuke Niizumi, Daiki Takeuchi, Yasunori Ohishi, Noboru Harada, Masahiro Yasuda, Shunsuke Tsubaki, Keisuke Imoto
251                                    </span>
252                                </p>
253                            </a>
254                            <a class="w3-text" href="fujita24_interspeech.html">
255                                <p>
256                                    Audio Fingerprinting with Holographic Reduced Representations
257                                    <br>
258                                    <span class="w3-text w3-text-theme">
259                                        Yusuke Fujita, Tatsuya Komatsu
260                                    </span>
261                                </p>
262                            </a>
263                            <a class="w3-text" href="meyer24b_interspeech.html">
264                                <p>
265                                    RAST: A Reference-Audio Synchronization Tool for Dubbed Content
266                                    <br>
267                                    <span class="w3-text w3-text-theme">
268                                        David Meyer, Eitan Abecassis, Clara Fernandez-Labrador, Christopher Schroers
269                                    </span>
270                                </p>
271                            </a>
272                            <a class="w3-text" href="li24ja_interspeech.html">
273                                <p>
274                                    YOLOPitch: A Time-Frequency Dual-Branch YOLO Model for Pitch Estimation
275                                    <br>
276                                    <span class="w3-text w3-text-theme">
277                                        Xuefei Li, Hao Huang, Ying Hu, Liang He, Jiabao Zhang, Yuyi Wang
278                                    </span>
279                                </p>
280                            </a>
281                            <a class="w3-text" href="ullah24_interspeech.html">
282                                <p>
283                                    Reduce, Reuse, Recycle: Is Perturbed Data Better than Other Language Augmentation for Low Resource Self-Supervised Speech Models
284                                    <br>
285                                    <span class="w3-text w3-text-theme">
286                                        Asad Ullah, Alessandro Ragano, Andrew Hines
287                                    </span>
288                                </p>
289                            </a>
290                            <a class="w3-text" href="pieper24_interspeech.html">
291                                <p>
292                                    AlignNet: Learning dataset score alignment functions to enable better training of speech quality estimators
293                                    <br>
294                                    <span class="w3-text w3-text-theme">
295                                        Jaden Pieper, Stephen Voran
296                                    </span>
297                                </p>
298                            </a>
299                        </div>
300                    </div>
301                    <br>
302                    <div class="w3-content" style="height:10px"  id="Acoustic Event Detection and Classification 2"></div>
303                    <div class="w3-card w3-round w3-white w3-padding">
304                        <div class="w3-container"  style="margin-top:40px">
305                            <h4 class="w3-center">Acoustic Event Detection and Classification 2</h4>
306                            <hr>
307                            <a class="w3-text" href="liang24_interspeech.html">
308                                <p>
309                                    Improving Audio Classification with Low-Sampled Microphone Input: An Empirical Study Using Model Self-Distillation
310                                    <br>
311                                    <span class="w3-text w3-text-theme">
312                                        Dawei Liang, Alice Zhang, David Harwath, Edison Thomaz
313                                    </span>
314                                </p>
315                            </a>
316                            <a class="w3-text" href="mu24_interspeech.html">
317                                <p>
318                                    MFF-EINV2: Multi-scale Feature Fusion across Spectral-Spatial
318-Temporal Domains for Sound Event Localization and Detection
319                                    <br>
320                                    <span class="w3-text w3-text-theme">
321                                        Da Mu, Zhicheng Zhang, Haobo Yue
322                                    </span>
323                                </p>
324                            </a>
325                            <a class="w3-text" href="nam24_interspeech.html">
326                                <p>
327                                    Diversifying and Expanding Frequency-Adaptive Convolution Kernels for Sound Event Detection
328                                    <br>
329                                    <span class="w3-text w3-text-theme">
330                                        Hyeonuk Nam, Seong-Hu Kim, Deokki Min, Junhyeok Lee, Yong-Hwa Park
331                                    </span>
332                                </p>
333                            </a>
334                            <a class="w3-text" href="ho24_interspeech.html">
335                                <p>
336                                    Stream-based Active Learning for Anomalous Sound Detection in Machine Condition Monitoring
337                                    <br>
338                                    <span class="w3-text w3-text-theme">
339                                        Tuan Vu Ho, Kota Dohi, Yohei Kawaguchi
340                                    </span>
341                                </p>
342                            </a>
343                            <a class="w3-text" href="jiang24c_interspeech.html">
344                                <p>
345                                    AnoPatch: Towards Better Consistency in Machine Anomalous Sound Detection
346                                    <br>
347                                    <span class="w3-text w3-text-theme">
348                                        Anbai Jiang, Bing Han, Zhiqiang Lv, Yufeng Deng, Wei-Qiang Zhang, Xie Chen, Yanmin Qian, Jia Liu, Pingyi Fan
349                                    </span>
350                                </p>
351                            </a>
352                            <a class="w3-text" href="xie24d_interspeech.html">
353                                <p>
354                                    FakeSound: Deepfake General Audio Detection
355                                    <br>
356                                    <span class="w3-text w3-text-theme">
357                                        Zeyu Xie, Baihan Li, Xuenan Xu, Zheng Liang, Kai Yu, Mengyue Wu
358                                    </span>
359                                </p>
360                            </a>
361                            <a class="w3-text" href="ghaffarzadegan24_interspeech.html">
362                                <p>
363                                    Sound of Traffic: A Dataset for Acoustic Traffic Identification and Counting
364                                    <br>
365                                    <span class="w3-text w3-text-theme">
366                                        Shabnam Ghaffarzadegan, Luca Bondi, Wei-Chang Lin, Abinaya Kumar, Ho-Hsiang Wu, Hans-Georg Horst, Samarjit Das
367                                    </span>
368                                </p>
369                            </a>
370                        </div>
371                    </div>
372                    <br>
373                    <div class="w3-content" style="height:10px"  id="Detection and Classification of Bioacoustic Signals"></div>
374                    <div class="w3-card w3-round w3-white w3-padding">
375                        <div class="w3-container"  style="margin-top:40px">
376                            <h4 class="w3-center">Detection and Classification of Bioacoustic Signals</h4>
377                            <hr>
378                            <a class="w3-text" href="kumar24_interspeech.html">
379                                <p>
380                                    Vision Transformer Segmentation for Visual Bird Sound Denoising
381                                    <br>
382                                    <span class="w3-text w3-text-theme">
383                                        Sahil Kumar, Jialu Li, Youshan Zhang
384                                    </span>
385                                </p>
386                            </a>
387                            <a class="w3-text" href="jing24_interspeech.html">
388                                <p>
389                                    DB3V: A Dialect Dominated Dataset of Bird Vocalisation for Cross-corpus Bird Species Recognition
390                                    <br>
391                                    <span class="w3-text w3-text-theme">
392                                        Xin Jing, Luyang Zhang, Jiangjian Xie, Alexander Gebhard, Alice Baird, Björn Schuller
393                                    </span>
394                                </p>
395                            </a>
396                            <a class="w3-text" href="cauzinille24_interspeech.html">
397                                <p>
398                                    Investigating self-supervised speech models' ability to classify animal vocalizations: The case of gibbon's vocal signatures
399                                    <br>
400                                    <span class="w3-text w3-text-theme">
401                                        Jules Cauzinille, Benoît Favre, Ricard Marxer, Dena Clink, Abdul Hamid Ahmad, Arnaud Rey
402                                    </span>
403                                </p>
404                            </a>
405                            <a class="w3-text" href="qiu24_interspeech.html">
406                                <p>
407                                    Study Selectively: An Adaptive Knowledge Distillation based on a Voting Network for Heart Sound Classification
408                                    <br>
409                                    <span class="w3-text w3-text-theme">
410                                        Xihang Qiu, Lixian Zhu, Zikai Song, Zeyu Chen, Haojie Zhang, Kun Qian, Ye Zhang, Bin Hu, Yoshiharu Yamamoto, Björn W. Schuller
411                                    </span>
412                                </p>
413                            </a>
414                            <a class="w3-text" href="lin24_interspeech.html">
415                                <p>
416                                    SimuSOE: A Simulated Snoring Dataset for Obstructive Sleep Apnea-Hypopnea Syndrome Evaluation during Wakefulness
417                                    <br>
418                                    <span class="w3-text w3-text-theme">
419                                        Jie Lin, Xiuping Yang, Li Xiao, Xinhong Li, Weiyan Yi, Yuhong Yang, Weiping Tu, Xiong Chen
420                                    </span>
421                                </p>
422                            </a>
423                        </div>
424                    </div>
425                    <br>
426                    <div class="w3-content" style="height:10px"  id="Acoustic Echo Cancellation"></div>
427                    <div class="w3-card w3-round w3-white w3-padding">
428                        <div class="w3-container"  style="margin-top:40px">
429                            <h4 class="w3-center">Acoustic Echo Cancellation</h4>
430                            <hr>
431                            <a class="w3-text" href="nayak24_interspeech.html">
432                                <p>
433                                    Multi-mic Echo Cancellation Coalesced with Beamforming for Real World Adverse Acoustic Conditions
434                                    <br>
435                                    <span class="w3-text w3-text-theme">
436                                        Premanand Nayak, Kamini Sabu, M. Ali Basha Shaik
437                                    </span>
438                                </p>
439                            </a>
440                            <a class="w3-text" href="khanagha24_interspeech.html">
441                                <p>
442                                    Interference Aware Training Target for DNN based joint Acoustic Echo Cancellation and Noise Suppression
443                                    <br>
444                                    <span class="w3-text w3-text-theme">
445                                        Vahid Khanagha, Dimitris Koutsaidis, Kaustubh Kalgaonkar, Sriram Srinivasan
446                                    </span>
447                                </p>
448                            </a>
449                            <a class="w3-text" href="gao24b_interspeech.html">
450                                <p>
451                                    Low Complexity Echo Delay Estimator Based on Binarized Feature Matching
452                                    <br>
453                                    <span class="w3-text w3-text-theme">
454                                        Yi Gao, Xiang Su
455                                    </span>
456                                </p>
457                            </a>
458                            <a class="w3-text" href="ni24_interspeech.html">
459                                <p>
460                                    MSA-DPCRN: A Multi-Scale Asymmetric Dual-Path Convolution Recurrent Network with Attentional Feature Fusion for Acoustic Echo Cancellation
461                                    <br>
462                                    <span class="w3-text w3-text-theme">
463                                        Ye Ni, Cong Pang, Chengwei Huang, Cairong Zou
464                                    </span>
465                                </p>
466                            </a>
467                            <a class="w3-text" href="schwartz24_interspeech.html">
468                                <p>
469                                    Efficient Joint Bemforming and Acoustic Echo Cancellation Structure for Conference Call Scenarios
470                                    <br>
471                                    <span class="w3-text w3-text-theme">
472                                        Ofer Schwartz, Sharon Gannot
473                                    </span>
474                                </p>
475                            </a>
476                            <a class="w3-text" href="zhao24b_interspeech.html">
477                                <p>
478                                    SDAEC: Signal Decoupling for Advancing Acoustic Echo Cancellation
479                                    <br>
480                                    <span class="w3-text w3-text-theme">
481                                        Fei Zhao, Jinjiang Liu, Xueliang Zhang
482                                    </span>
483                                </p>
484                            </a>
485                        </div>
486                    </div>
487                    <br>
488                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Voice Conversion 1"></div>
489                    <div class="w3-card w3-round w3-white w3-padding">
490                        <div class="w3-container"  style="margin-top:40px">
491                            <h4 class="w3-center">Speech Synthesis: Voice Conversion 1</h4>
492                            <hr>
493                            <a class="w3-text" href="seki24_interspeech.html">
494                                <p>
495                                    Spatial Voice Conversion: Voice Conversion Preserving Spatial Information and Non-target Signals
496                                    <br>
497                                    <span class="w3-text w3-text-theme">
498                                        Kentaro Seki, Shinnosuke Takamichi, Norihiro Takamune, Yuki Saito, Kanami Imamura, Hiroshi Saruwatari
499                                    </span>
500                                </p>
501                            </a>
502                            <a class="w3-text" href="baade24_interspeech.html">
503                                <p>
504                                    Neural Codec Language Models for Disentangled and Textless Voice Conversion
505                                    <br>
506                                    <span class="w3-text w3-text-theme">
507                                        Alan Baade, Puyuan Peng, David Harwath
508                                    </span>
509                                </p>
510                            </a>
511                            <a class="w3-text" href="morrison24_interspeech.html">
512                                <p>
513                                    Fine-Grained and Interpretable Neural Speech Editing
514                                    <br>
515                                    <span class="w3-text w3-text-theme">
516                                        Max Morrison, Cameron Churchwell, Nathan Pruyne, Bryan Pardo
517                                    </span>
518                                </p>
519                            </a>
520                            <a class="w3-text" href="kaneko24_interspeech.html">
521                                <p>
522                                    FastVoiceGrad: One-step Diffusion-Based Voice Conversion with Adversarial Conditional Diffusion Distillation
523                                    <br>
524                                    <span class="w3-text w3-text-theme">
525                                        Takuhiro Kaneko, Hirokazu Kameoka, Kou Tanaka, Yuto Kondo
526                                    </span>
527                                </p>
528                            </a>
529                            <a class="w3-text" href="ning24_interspeech.html">
530                                <p>
531                                    DualVC 3: Leveraging Language Model Generated Pseudo Context for End-to-end Low Latency Streaming Voice Conversion
532                                    <br>
533                                    <span class="w3-text w3-text-theme">
534                                        Ziqian Ning, Shuai Wang, Pengcheng Zhu, Zhichao Wang, Jixun Yao, Lei Xie, Mengxiao Bi
535                                    </span>
536                                </p>
537                            </a>
538                            <a class="w3-text" href="qi24_interspeech.html">
539                                <p>
540                                    Towards Realistic Emotional Voice Conversion using Controllable Emotional Intensity
541                                    <br>
542                                    <span class="w3-text w3-text-theme">
543                                        Tianhua Qi, Shiyan Wang, Cheng Lu, Yan Zhao, Yuan Zong, Wenming Zheng
544                                    </span>
545                                </p>
546                            </a>
547                        </div>
548                    </div>
549                    <br>
550                    <div class="w3-content" style="height:10px"  id="Neural Network Architectures for ASR 2"></div>
551                    <div class="w3-card w3-round w3-white w3-padding">
552                        <div class="w3-container"  style="margin-top:40px">
553                            <h4 class="w3-center">Neural Network Architectures for ASR 2</h4>
554                            <hr>
555                            <a class="w3-text" href="nakagome24_interspeech.html">
556                                <p>
557                                    InterBiasing: Boost Unseen Word Recognition through Biasing Intermediate Predictions
558                                    <br>
559                                    <span class="w3-text w3-text-theme">
560                                        Yu Nakagome, Michael Hentschel
561                                    </span>
562                                </p>
563                            </a>
564                            <a class="w3-text" href="meng24_interspeech.html">
565                                <p>
566                                    SEQ-former: A context-enhanced and efficient automatic speech recognition framework
567                                    <br>
568                                    <span class="w3-text w3-text-theme">
569                                        Qinglin Meng, Min Liu, Kaixun Huang, Kun Wei, Lei Xie, Zongfeng Quan, Weihong Deng, Quan Lu, Ning Jiang, Guoqing Zhao
570                                    </span>
571                                </p>
572                            </a>
573                            <a class="w3-text" href="flynn24b_interspeech.html">
574                                <p>
575                                    How Much Context Does My Attention-Based ASR System Need?
576                                    <br>
577                                    <span class="w3-text w3-text-theme">
578                                        Robert Flynn, Anton Ragni
579                                    </span>
580                                </p>
581                            </a>
582                            <a class="w3-text" href="vitale24_interspeech.html">
583                                <p>
584                                    Rich speech signal: exploring and exploiting  end-to-end automatic speech recognizers’ ability to model hesitation phenomena
585                                    <br>
586                                    <span class="w3-text w3-text-theme">
587                                        Vincenzo Norman Vitale, Loredana Schettino, Francesco Cutugno
588                                    </span>
589                                </p>
590                            </a>
591                            <a class="w3-text" href="zhang24q_interspeech.html">
592                                <p>
593                                    Transmitted and Aggregated Self-Attention for Automatic Speech Recognition
594                                    <br>
595                                    <span class="w3-text w3-text-theme">
596                                        Tian-Hao Zhang, Xinyuan Qian, Feng Chen, Xu-Cheng Yin
597                                    </span>
598                                </p>
599                            </a>
600                            <a class="w3-text" href="prabhu24_interspeech.html">
601                                <p>
602                                    MULTI-CONVFORMER: Extending Conformer with Multiple Convolution Kernels
603                                    <br>
604                                    <span class="w3-text w3-text-theme">
605                                        Darshan Prabhu, Yifan Peng, Preethi Jyothi, Shinji Watanabe
606                                    </span>
607                                </p>
608                            </a>
609                            <a class="w3-text" href="miyazaki24_interspeech.html">
610                                <p>
611                                    Exploring the Capability of Mamba in Speech Applications
612                                    <br>
613                                    <span class="w3-text w3-text-theme">
614                                        Koichi Miyazaki, Yoshiki Masuyama, Masato Murata
615                                    </span>
616                                </p>
617                            </a>
618                            <a class="w3-text" href="wan24_interspeech.html">
619                                <p>
620                                    Lightweight Transducer Based on Frame-Level Criterion
621                                    <br>
622                                    <span class="w3-text w3-text-theme">
623                                        Genshun Wan, Mengzhi Wang, Tingzhi Mao, Hang Chen, Zhongfu Ye
624                                    </span>
625                                </p>
626                            </a>
627                            <a class="w3-text" href="gupta24_interspeech.html">
628                                <p>
629                                    Exploring the limits of decoder-only models trained on public speech recognition corpora
630                                    <br>
631                                    <span class="w3-text w3-text-theme">
632                                        Ankit Gupta, George Saon, Brian Kingsbury
633                                    </span>
634                                </p>
635                            </a>
636                            <a class="w3-text" href="gong24b_interspeech.html">
637                                <p>
638                                    Contextual Biasing Speech Recognition in Speech-enhanced Large Language Model
639                                    <br>
640                                    <span class="w3-text w3-text-theme">
641                                        Xun Gong, Anqi Lv, Zhiming Wang, Yanmin Qian
642                                    </span>
643                                </p>
644                            </a>
645                        </div>
646                    </div>
647                    <br>
648                    <div class="w3-content" style="height:10px"  id="Decoding Algorithms"></div>
649                    <div class="w3-card w3-round w3-white w3-padding">
650                        <div class="w3-container"  style="margin-top:40px">
651                            <h4 class="w3-center">Decoding Algorithms</h4>
652                            <hr>
653                            <a class="w3-text" href="wang24k_interspeech.html">
654                                <p>
655                                    Towards Effective and Efficient Non-autoregressive Decoding Using Block-based Attention Mask
656                                    <br>
657                                    <span class="w3-text w3-text-theme">
658                                        Tianzi Wang, Xurong Xie, Zhaoqing Li, Shoukang Hu, Zengrui Jin, Jiajun Deng, Mingyu Cui, Shujie Hu, Mengzhe Geng, Guinan Li, Helen Meng, Xunying Liu
659                                    </span>
660                                </p>
661                            </a>
662                            <a class="w3-text" href="zou24_interspeech.html">
663                                <p>
664                                    E-Paraformer: A Faster and Better  Parallel Transformer for Non-autoregressive End-to-End Mandarin Speech Recognition
665                                    <br>
666                                    <span class="w3-text w3-text-theme">
667                                        Kun Zou, Fengyun Tan, Ziyang Zhuang, Chenfeng Miao, Tao Wei, Shaodan Zhai, Zijian Li, Wei Hu, Shaojun Wang, Jing Xiao
668                                    </span>
669                                </p>
670                            </a>
671                            <a class="w3-text" href="ciaperoni24_interspeech.html">
672                                <p>
673                                    Beam-search SIEVE for low-memory speech recognition
674                                    <br>
675                                    <span class="w3-text w3-text-theme">
676                                        Martino Ciaperoni, Athanasios Katsamanis, Aristides Gionis, Panagiotis Karras
677                                    </span>
678                                </p>
679                            </a>
680                            <a class="w3-text" href="galvez24_interspeech.html">
681                                <p>
682                                    Speed of Light Exact Greedy Decoding for RNN-T Speech Recognition Models on GPU
683                                    <br>
684                                    <span class="w3-text w3-text-theme">
685                                        Daniel Galvez, Vladimir Bataev, Hainan Xu, Tim Kaldewey
686                                    </span>
687                                </p>
688                            </a>
689                            <a class="w3-text" href="wang24w_interspeech.html">
690                                <p>
691                                    Contextual Biasing with the Knuth-Morris-Pratt Matching Algorithm
692                                    <br>
693                                    <span class="w3-text w3-text-theme">
694                                        Weiran Wang, Zelin Wu, Diamantino Caseiro, Tsendsuren Munkhdalai, Khe Chai Sim, Pat Rondon, Golan Pundak, Gan Song, Rohit Prabhavalkar, Zhong Meng, Ding Zhao, Tara Sainath, Yanzhang He, Pedro Moreno Mengibar
695                                    </span>
696                                </p>
697                            </a>
698                            <a class="w3-text" href="takagi24_interspeech.html">
699                                <p>
700                                    Text-only Domain Adaptation for CTC-based Speech Recognition through Substitution of Implicit Linguistic Information in the Search Space
701                                    <br>
702                                    <span class="w3-text w3-text-theme">
703                                        Tatsunari Takagi, Yukoh Wakabayashi, Atsunori Ogawa, Norihide Kitaoka
704                                    </span>
705                                </p>
706                            </a>
707                        </div>
708                    </div>
709                    <br>
710                    <div class="w3-content" style="height:10px"  id="Pronunciation Assessment"></div>
711                    <div class="w3-card w3-round w3-white w3-padding">
712                        <div class="w3-container"  style="margin-top:40px">
713                            <h4 class="w3-center">Pronunciation Assessment</h4>
714                            <hr>
715                            <a class="w3-text" href="wang24la_interspeech.html">
716                                <p>
717                                    Pitch-Aware RNN-T for Mandarin Chinese Mispronunciation Detection and Diagnosis
718                                    <br>
719                                    <span class="w3-text w3-text-theme">
720                                        Xintong Wang, Mingqian Shi, Ye Wang
721                                    </span>
722                                </p>
723                            </a>
724                            <a class="w3-text" href="chen24c_interspeech.html">
725                                <p>
726                                    MultiPA: A Multi-task Speech Pronunciation Assessment Model for Open Response Scenarios
727                                    <br>
728                                    <span class="w3-text w3-text-theme">
729                                        Yu-Wen Chen, Zhou Yu, Julia Hirschberg
730                                    </span>
731                                </p>
732                            </a>
733                            <a class="w3-text" href="cao24b_interspeech.html">
734                                <p>
735                                    A Framework for Phoneme-Level Pronunciation Assessment Using CTC
736                                    <br>
737                                    <span class="w3-text w3-text-theme">
738                                        Xinwei Cao, Zijian Fan, Torbjørn Svendsen, Giampiero Salvi
739                                    </span>
740                                </p>
741                            </a>
742                            <a class="w3-text" href="shahin24_interspeech.html">
743                                <p>
744                                    Phonological-Level Mispronunciation Detection and Diagnosis
745                                    <br>
746                                    <span class="w3-text w3-text-theme">
747                                        Mostafa Shahin, Beena Ahmed
748                                    </span>
749                                </p>
750                            </a>
751                            <a class="w3-text" href="do24_interspeech.html">
752                                <p>
753                                    Acoustic Feature Mixup for Balanced Multi-aspect Pronunciation Assessment
754                                    <br>
755                                    <span class="w3-text w3-text-theme">
756                                        Heejin Do, Wonjun Lee, Gary Geunbae Lee
757                                    </span>
758                                </p>
759                            </a>
760                            <a class="w3-text" href="phan24_interspeech.html">
761                                <p>
762                                    Automated content assessment and feedback for Finnish L2 learners in a picture description speaking task
763                                    <br>
764                                    <span class="w3-text w3-text-theme">
765                                        Nhan Phan, Anna von Zansen, Maria Kautonen, Ekaterina Voskoboinik, Tamas Grosz, Raili Hilden, Mikko Kurimo
766                                    </span>
767                                </p>
768                            </a>
769                        </div>
770                    </div>
771                    <br>
772                    <div class="w3-content" style="height:10px"  id="Spoken Language Processing"></div>
773                    <div class="w3-card w3-round w3-white w3-padding">
774                        <div class="w3-container"  style="margin-top:40px">
775                            <h4 class="w3-center">Spoken Language Processing</h4>
776                            <hr>
777                            <a class="w3-text" href="wang24c_interspeech.html">
778                                <p>
779                                    Query-by-Example Keyword Spotting Using Spectral-Temporal Graph Attentive Pooling and Multi-Task Learning
780                                    <br>
781                                    <span class="w3-text w3-text-theme">
782                                        Zhenyu Wang, Shuyu Kong, Li Wan, Biqiao Zhang, Yiteng Huang, Mumin Jin, Ming Sun, Xin Lei, Zhaojun Yang
783                                    </span>
784                                </p>
785                            </a>
786                            <a class="w3-text" href="jung24_interspeech.html">
787                                <p>
788                                    Relational Proxy Loss for Audio-Text based Keyword Spotting
789                                    <br>
790                                    <span class="w3-text w3-text-theme">
791                                        Youngmoon Jung, Seungjin Lee, Joon-Young Yang, Jaeyoung Roh, Chang Woo Han, Hoon-Young Cho
792                                    </span>
793                                </p>
794                            </a>
795                            <a class="w3-text" href="jin24d_interspeech.html">
796                                <p>
797                                    CTC-aligned Audio-Text Embedding for Streaming Open-vocabulary Keyword Spotting
798                                    <br>
799                                    <span class="w3-text w3-text-theme">
800                                        Sichen Jin, Youngmoon Jung, Seungjin Lee, Jaeyoung Roh, Changwoo Han, Hoonyoung Cho
801                                    </span>
802                                </p>
803                            </a>
804                            <a class="w3-text" href="li24r_interspeech.html">
805                                <p>
806                                    Text-aware Speech Separation for Multi-talker Keyword Spotting
807                                    <br>
808                                    <span class="w3-text w3-text-theme">
809                                        Haoyu Li, Baochen Yang, Yu Xi, Linfeng Yu, Tian Tan, Hao Li, Kai Yu
810                                    </span>
811                                </p>
812                            </a>
813                            <a class="w3-text" href="yen24_interspeech.html">
814                                <p>
815                                    Language-Universal Speech Attributes Modeling for Zero-Shot Multilingual Spoken Keyword Recognition
816                                    <br>
817                                    <span class="w3-text w3-text-theme">
818                                        Hao Yen, Pin-Jui Ku, Sabato Marco Siniscalchi, Chin-Hui Lee
819                                    </span>
820                                </p>
821                            </a>
822                            <a class="w3-text" href="monteiro24_interspeech.html">
823                                <p>
824                                    Adding User Feedback To Enhance CB-Whisper
825                                    <br>
826                                    <span class="w3-text w3-text-theme">
827                                        Raul Monteiro
828                                    </span>
829                                </p>
830                            </a>
831                            <a class="w3-text" href="peng24b_interspeech.html">
832                                <p>
833                                    OWSM v3.1: Better and Faster Open Whisper-Style Speech Models based on E-Branchformer
834                                    <br>
835                                    <span class="w3-text w3-text-theme">
836                                        Yifan Peng, Jinchuan Tian, William Chen, Siddhant Arora, Brian Yan, Yui Sudo, Muhammad Shakeel, Kwanghee Choi, Jiatong Shi, Xuankai Chang, Jee-weon Jung, Shinji Watanabe
837                                    </span>
838                                </p>
839                            </a>
840                        </div>
841                    </div>
842                    <br>
843                    <div class="w3-content" style="height:10px"  id="Spoken Machine Translation 2"></div>
844                    <div class="w3-card w3-round w3-white w3-padding">
845                        <div class="w3-container"  style="margin-top:40px">
846                            <h4 class="w3-center">Spoken Machine Translation 2</h4>
847                            <hr>
848                            <a class="w3-text" href="chen24m_interspeech.html">
849                                <p>
850                                    Parameter-Efficient Adapter Based on Pre-trained Models for Speech Translation
851                                    <br>
852                                    <span class="w3-text w3-text-theme">
853                                        Nan Chen, Yonghe Wang, Feilong Bao
854                                    </span>
855                                </p>
856                            </a>
857                            <a class="w3-text" href="abdullah24_interspeech.html">
858                                <p>
859                                    Wave to Interlingua: Analyzing Representations of Multilingual Speech Transformers for Spoken Language Translation
860                                    <br>
861                                    <span class="w3-text w3-text-theme">
862                                        Badr M. Abdullah, Mohammed Maqsood Shaik, Dietrich Klakow
863                                    </span>
864                                </p>
865                            </a>
866                            <a class="w3-text" href="chen24w_interspeech.html">
867                                <p>
868                                    Knowledge-Preserving Pluggable Modules for Multilingual Speech Translation Tasks
869                                    <br>
870                                    <span class="w3-text w3-text-theme">
871                                        Nan Chen, Yonghe Wang, Feilong Bao
872                                    </span>
873                                </p>
874                            </a>
875                            <a class="w3-text" href="rabatin24_interspeech.html">
876                                <p>
877                                    Navigating the Minefield of MT Beam Search in Cascaded Streaming Speech Translation
878                                    <br>
879                                    <span class="w3-text w3-text-theme">
880                                        Rastislav Rabatin, Frank Seide, Ernie Chang
881                                    </span>
882                                </p>
883                            </a>
884                            <a class="w3-text" href="wang24aa_interspeech.html">
885                                <p>
886                                    Soft Language Identification for Language-Agnostic Many-to-One End-to-End Speech Translation
887                                    <br>
888                                    <span class="w3-text w3-text-theme">
889                                        Peidong Wang, Jian Xue, Jinyu Li, Junkun Chen, Aswin Shanmugam Subramanian
890                                    </span>
891                                </p>
892                            </a>
893                            <a class="w3-text" href="oneata24_interspeech.html">
894                                <p>
895                                    Translating speech with just images
896                                    <br>
897                                    <span class="w3-text w3-text-theme">
898                                        Dan Oneata, Herman Kamper
899                                    </span>
900                                </p>
901                            </a>
902                            <a class="w3-text" href="khurana24_interspeech.html">
903                                <p>
904                                    ZeroST: Zero-Shot Speech Translation
905                                    <br>
906                                    <span class="w3-text w3-text-theme">
907                                        Sameer Khurana, Chiori Hori, Antoine Laurent, Gordon Wichern, Jonathan Le Roux
908                                    </span>
909                                </p>
910                            </a>
911                        </div>
912                    </div>
913                    <br>
914                    <div class="w3-content" style="height:10px"  id="Biosignal-enabled Spoken Communication"></div>
915                    <div class="w3-card w3-round w3-white w3-padding">
916                        <div class="w3-container"  style="margin-top:40px">
917                            <h4 class="w3-center">Biosignal-enabled Spoken Communication</h4>
918                            <hr>
919                            <a class="w3-text" href="li24ca_interspeech.html">
920                                <p>
921                                    A multimodal approach to study the nature of coordinative patterns underlying speech rhythm
922                                    <br>
923                                    <span class="w3-text w3-text-theme">
924                                        Jinyu Li, Leonardo Lancia
925                                    </span>
926                                </p>
927                            </a>
928                            <a class="w3-text" href="wu24k_interspeech.html">
929                                <p>
930                                    Towards EMG-to-Speech with Necklace Form Factor
931                                    <br>
932                                    <span class="w3-text w3-text-theme">
933                                        Peter Wu, Ryan Kaveh, Raghav Nautiyal, Christine Zhang, Albert Guo, Anvitha Kachinthaya, Tavish Mishra, Bohan Yu, Alan W Black, Rikky Muller, Gopala Krishna Anumanchipalli
934                                    </span>
935                                </p>
936                            </a>
937                            <a class="w3-text" href="bras24_interspeech.html">
938                                <p>
939                                    Using articulated speech EEG signals for imagined speech decoding
940                                    <br>
941                                    <span class="w3-text w3-text-theme">
942                                        Chris Bras, Tanvina Patel, Odette Scharenborg
943                                    </span>
944                                </p>
945                            </a>
946                            <a class="w3-text" href="kwon24_interspeech.html">
947                                <p>
948                                    Direct Speech Synthesis from Non-Invasive, Neuromagnetic Signals
949                                    <br>
950                                    <span class="w3-text w3-text-theme">
951                                        Jinuk Kwon, David Harwath, Debadatta Dash, Paul Ferrari, Jun Wang
952                                    </span>
953                                </p>
954                            </a>
955                            <a class="w3-text" href="yang24o_interspeech.html">
956                                <p>
957                                    Optical Flow Guided Tongue Trajectory Generation for Diffusion-based Acoustic to Articulatory Inversion
958                                    <br>
959                                    <span class="w3-text w3-text-theme">
960                                        Yudong Yang, Rongfeng Su, Rukiye Ruzi, Manwa Ng, Shaofeng Zhao, Nan Yan, Lan Wang
961                                    </span>
962                                </p>
963                            </a>
964                            <a class="w3-text" href="jain24_interspeech.html">
965                                <p>
966                                    Multimodal Segmentation for Vocal Tract Modeling
967                                    <br>
968                                    <span class="w3-text w3-text-theme">
969                                        Rishi Jain, Bohan Yu, Peter Wu, Tejas Prabhune, Gopala Anumanchipalli
970                                    </span>
971                                </p>
972                            </a>
973                            <a class="w3-text" href="bandekar24_interspeech.html">
974                                <p>
975                                    Articulatory synthesis using representations learnt through phonetic label-aware contrastive loss
976                                    <br>
977                                    <span class="w3-text w3-text-theme">
978                                        Jesuraj Bandekar, Sathvik Udupa, Prasanta Kumar Ghosh
979                                    </span>
980                                </p>
981                            </a>
982                            <a class="w3-text" href="yan24b_interspeech.html">
983                                <p>
984                                    Auditory Attention Decoding in Four-Talker Environment with EEG
985                                    <br>
986                                    <span class="w3-text w3-text-theme">
987                                        Yujie Yan, Xiran Xu, Haolin Zhu, Pei Tian, Zhongshu Ge, Xihong Wu, Jing Chen
988                                    </span>
989                                </p>
990                            </a>
991                            <a class="w3-text" href="lin24f_interspeech.html">
992                                <p>
993                                    ASA: An Auditory Spatial Attention Dataset with Multiple Speaking Locations
994                                    <br>
995                                    <span class="w3-text w3-text-theme">
996                                        Zijie Lin, Tianyu He, Siqi Cai, Haizhou Li
997                                    </span>
998                                </p>
999                            </a>
1000                            <a class="w3-text" href="pahuja24_interspeech.html">
1001                                <p>
1002                                    Leveraging Graphic and Convolutional Neural Networks for Auditory Attention Detection with EEG
1003                                    <br>
1004                                    <span class="w3-text w3-text-theme">
1005                                        Saurav Pahuja, Gabriel Ivucic, Pascal Himmelmann, Siqi Cai, Tanja Schultz, Haizhou Li
1006                                    </span>
1007                                </p>
1008                            </a>
1009                        </div>
1010                    </div>
1011                    <br>
1012                    <div class="w3-content" style="height:10px"  id="Individual and Social Factors in Phonetics"></div>
1013                    <div class="w3-card w3-round w3-white w3-padding">
1014                        <div class="w3-container"  style="margin-top:40px">
1015                            <h4 class="w3-center">Individual and Social Factors in Phonetics</h4>
1016                            <hr>
1017                            <a class="w3-text" href="pistor24_interspeech.html">
1018                                <p>
1019                                    Echoes of Implicit Bias Exploring Aesthetics and Social Meanings of Swiss German Dialect Features
1020                                    <br>
1021                                    <span class="w3-text w3-text-theme">
1022                                        Tillmann Pistor, Adrian Leemann
1023                                    </span>
1024                                </p>
1025                            </a>
1026                            <a class="w3-text" href="li24ra_interspeech.html">
1027                                <p>
1028                                    In search of structure and correspondence in intra-speaker trial-to-trial variability
1029                                    <br>
1030                                    <span class="w3-text w3-text-theme">
1031                                        Vivian G. Li
1032                                    </span>
1033                                </p>
1034                            </a>
1035                            <a class="w3-text" href="smith24_interspeech.html">
1036                                <p>
1037                                    Modelled Multivariate Overlap: A method for measuring vowel merger
1038                                    <br>
1039                                    <span class="w3-text w3-text-theme">
1040                                        Irene Smith, Morgan Sonderegger, The Spade Consortium
1041                                    </span>
1042                                </p>
1043                            </a>
1044                            <a class="w3-text" href="ochi24_interspeech.html">
1045                                <p>
1046                                    Entrainment Analysis and Prosody Prediction of Subsequent Interlocutor’s Backchannels in Dialogue
1047                                    <br>
1048                                    <span class="w3-text w3-text-theme">
1049                                        Keiko Ochi, Koji Inoue, Divesh Lala, Tatsuya Kawahara
1050                                    </span>
1051                                </p>
1052                            </a>
1053                            <a class="w3-text" href="tanner24_interspeech.html">
1054                                <p>
1055                                    Exploring the anatomy of articulation rate in spontaneous English speech: relationships between utterance length effects and social factors
1056                                    <br>
1057                                    <span class="w3-text w3-text-theme">
1058                                        James Tanner, Morgan Sonderegger, Jane Stuart-Smith, Tyler Kendall, Jeff Mielke, Robin Dodsworth, Erik Thomas
1059                                    </span>
1060                                </p>
1061                            </a>
1062                            <a class="w3-text" href="taylor24_interspeech.html">
1063                                <p>
1064                                    Familiar and Unfamiliar Speaker Identification in Speech and Singing
1065                                    <br>
1066                                    <span class="w3-text w3-text-theme">
1067                                        Katelyn Taylor, Amelia Gully, Helena Daffern
1068                                    </span>
1069                                </p>
1070                            </a>
1071                        </div>
1072                    </div>
1073                    <br>
1074                    <div class="w3-content" style="height:10px"  id="Paralinguistics"></div>
1075                    <div class="w3-card w3-round w3-white w3-padding">
1076                        <div class="w3-container"  style="margin-top:40px">
1077                            <h4 class="w3-center">Paralinguistics</h4>
1078                            <hr>
1079                            <a class="w3-text" href="parragallego24_interspeech.html">
1080                                <p>
1081                                    Cross-transfer Knowledge between Speech and Text Encoders to Evaluate Customer Satisfaction
1082                                    <br>
1083                                    <span class="w3-text w3-text-theme">
1084                                        Luis Felipe Parra-Gallego, Tilak Purohit, Bogdan Vlasenko, Juan Rafael Orozco-Arroyave, Mathew Magimai.-Doss
1085                                    </span>
1086                                </p>
1087                            </a>
1088                            <a class="w3-text" href="kodali24_interspeech.html">
1089                                <p>
1090                                    Fine-tuning of Pre-trained Models for Classification of Vocal Intensity Category from Speech Signals
1091                                    <br>
1092                                    <span class="w3-text w3-text-theme">
1093                                        Manila Kodali, Sudarsana Reddy Kadiri, Paavo Alku
1094                                    </span>
1095                                </p>
1096                            </a>
1097                            <a class="w3-text" href="kathan24_interspeech.html">
1098                                <p>
1099                                    Real-world PTSD Recognition: A Cross-corpus and Cross-linguistic Evaluation
1100                                    <br>
1101                                    <span class="w3-text w3-text-theme">
1102                                        Alexander Kathan, Martin Bürger, Andreas Triantafyllopoulos, Sabrina Milkus, Jonas Hohmann, Pauline Muderlak, Jürgen Schottdorf, Richard Musil, Björn Schuller, Shahin Amiriparian
1103                                    </span>
1104                                </p>
1105                            </a>
1106                            <a class="w3-text" href="bhattacharya24_interspeech.html">
1107                                <p>
1108                                    Switching Tongues, Sharing Hearts: Identifying the Relationship between Empathy and Code-switching in Speech
1109                                    <br>
1110                                    <span class="w3-text w3-text-theme">
1111                                        Debasmita Bhattacharya, Eleanor Lin, Run Chen, Julia Hirschberg
1112                                    </span>
1113                                </p>
1114                            </a>
1115                        </div>
1116                    </div>
1117                    <br>
1118                    <div class="w3-content" style="height:10px"  id="Speaker Recognition: Adversarial and Spoofing Attacks"></div>
1119                    <div class="w3-card w3-round w3-white w3-padding">
1120                        <div class="w3-container"  style="margin-top:40px">
1121                            <h4 class="w3-center">Speaker Recognition: Adversarial and Spoofing Attacks</h4>
1122                            <hr>
1123                            <a class="w3-text" href="rosello24_interspeech.html">
1124                                <p>
1125                                    Anti-spoofing Ensembling Model: Dynamic Weight Allocation in Ensemble Models for Improved Voice Biometrics Security
1126                                    <br>
1127                                    <span class="w3-text w3-text-theme">
1128                                        Eros Rosello, Angel M. Gomez, Iván López-Espejo, Antonio M. Peinado, Juan M. Martín-Doñas
1129                                    </span>
1130                                </p>
1131                            </a>
1132                            <a class="w3-text" href="zhang24j_interspeech.html">
1133                                <p>
1134                                    Spoof Diarization: &quot;What Spoofed When&quot; in Partially Spoofed Audio
1135                                    <br>
1136                                    <span class="w3-text w3-text-theme">
1137                                        Lin Zhang, Xin Wang, Erica Cooper, Mireia Diez, Federico Landini, Nicholas Evans, Junichi Yamagishi
1138                                    </span>
1139                                </p>
1140                            </a>
1141                            <a class="w3-text" href="wu24b_interspeech.html">
1142                                <p>
1143                                    Spoofing Speech Detection by Modeling Local Spectro-Temporal and Long-term Dependency
1144                                    <br>
1145                                    <span class="w3-text w3-text-theme">
1146                                        Haochen Wu, Wu Guo, Zhentao Zhang, Wenting Zhao, Shengyu Peng, Jie Zhang
1147                                    </span>
1148                                </p>
1149                            </a>
1150                            <a class="w3-text" href="lu24b_interspeech.html">
1151                                <p>
1152                                    Improving Copy-Synthesis Anti-Spoofing Training Method with Rhythm and Speaker Perturbation
1153                                    <br>
1154                                    <span class="w3-text w3-text-theme">
1155                                        Jingze Lu, Yuxiang Zhang, Zhuo Li, Zengqiang Shang, Wenchao Wang, Pengyuan Zhang
1156                                    </span>
1157                                </p>
1158                            </a>
1159                            <a class="w3-text" href="kan24_interspeech.html">
1160                                <p>
1161                                    VoiceDefense: Protecting Automatic Speaker Verification Models Against Black-box Adversarial Attacks
1162                                    <br>
1163                                    <span class="w3-text w3-text-theme">
1164                                        Yip Keng Kan, Ke Xu, Hao Li, Jie Shi
1165                                    </span>
1166                                </p>
1167                            </a>
1168                            <a class="w3-text" href="chen24p_interspeech.html">
1169                                <p>
1170                                    Neural Codec-based Adversarial Sample Detection for Speaker Verification
1171                                    <br>
1172                                    <span class="w3-text w3-text-theme">
1173                                        Xuanjun Chen, Jiawei Du, Haibin Wu, Jyh-Shing Roger Jang, Hung-yi Lee
1174                                    </span>
1175                                </p>
1176                            </a>
1177                            <a class="w3-text" href="chen24_interspeech.html">
1178                                <p>
1179                                    Textual-Driven Adversarial Purification for Speaker Verification
1180                                    <br>
1181                                    <span class="w3-text w3-text-theme">
1182                                        Sizhou Chen, Yibo Bai, Jiadi Yao, Xiao-Lei Zhang, Xuelong Li
1183                                    </span>
1184                                </p>
1185                            </a>
1186                            <a class="w3-text" href="li24g_interspeech.html">
1187                                <p>
1188                                    Boosting the Transferability of Adversarial Examples with Gradient-Aligned Ensemble Attack for Speaker Recognition
1189                                    <br>
1190                                    <span class="w3-text w3-text-theme">
1191                                        Zhuhai Li, Jie Zhang, Wu Guo, Haochen Wu
1192                                    </span>
1193                                </p>
1194                            </a>
1195                            <a class="w3-text" href="truong24b_interspeech.html">
1196                                <p>
1197                                    Temporal-Channel Modeling in Multi-head Self-Attention for Synthetic Speech Detection
1198                                    <br>
1199                                    <span class="w3-text w3-text-theme">
1200                                        Duc-Tuan Truong, Ruijie Tao, Tuan Nguyen, Hieu-Thi Luong, Kong Aik Lee, Eng Siong Chng
1201                                    </span>
1202                                </p>
1203                            </a>
1204                        </div>
1205                    </div>
1206                    <br>
1207                    <div class="w3-content" style="height:10px"  id="Audio Event Detection and Classification 1"></div>
1208                    <div class="w3-card w3-round w3-white w3-padding">
1209                        <div class="w3-container"  style="margin-top:40px">
1210                            <h4 class="w3-center">Audio Event Detection and Classification 1</h4>
1211                            <hr>
1212                            <a class="w3-text" href="feng24b_interspeech.html">
1213                                <p>
1214                                    Can Synthetic Audio From Generative Foundation Models Assist Audio Recognition and Speech Modeling?
1215                                    <br>
1216                                    <span class="w3-text w3-text-theme">
1217                                        Tiantian Feng, Dimitrios Dimitriadis, Shrikanth S. Narayanan
1218                                    </span>
1219                                </p>
1220                            </a>
1221                            <a class="w3-text" href="dinkel24b_interspeech.html">
1222                                <p>
1223                                    Scaling up masked audio encoder learning for general audio classification
1224                                    <br>
1225                                    <span class="w3-text w3-text-theme">
1226                                        Heinrich Dinkel, Zhiyong Yan, Yongqing Wang, Junbo Zhang, Yujun Wang, Bin Wang
1227                                    </span>
1228                                </p>
1229                            </a>
1230                            <a class="w3-text" href="yadav24_interspeech.html">
1231                                <p>
1232                                    Audio Mamba: Selective State Spaces for Self-Supervised Audio Representations
1233                                    <br>
1234                                    <span class="w3-text w3-text-theme">
1235                                        Sarthak Yadav, Zheng-Hua Tan
1236                                    </span>
1237                                </p>
1238                            </a>
1239                            <a class="w3-text" href="cai24_interspeech.html">
1240                                <p>
1241                                    MAT-SED: A Masked Audio Transformer with Masked-Reconstruction Based Pre-training for Sound Event Detection
1242                                    <br>
1243                                    <span class="w3-text w3-text-theme">
1244                                        Pengfei Cai, Yan Song, Kang Li, Haoyu Song, Ian McLoughlin
1245                                    </span>
1246                                </p>
1247                            </a>
1248                            <a class="w3-text" href="ebbers24_interspeech.html">
1249                                <p>
1250                                    Sound Event Bounding Boxes
1251                                    <br>
1252                                    <span class="w3-text w3-text-theme">
1253                                        Janek Ebbers, François G. Germain, Gordon Wichern, Jonathan Le Roux
1254                                    </span>
1255                                </p>
1256                            </a>
1257                            <a class="w3-text" href="li24k_interspeech.html">
1258                                <p>
1259                                    Low-Complexity Acoustic Scene Classification Using Parallel Attention-Convolution Network
1260                                    <br>
1261                                    <span class="w3-text w3-text-theme">
1262                                        Yanxiong Li, Jiaxin Tan, Guoqing Chen, Jialong Li, Yongjie Si, Qianhua He
1263                                    </span>
1264                                </p>
1265                            </a>
1266                        </div>
1267                    </div>
1268                    <br>
1269                    <div class="w3-content" style="height:10px"  id="Source Separation 2"></div>
1270                    <div class="w3-card w3-round w3-white w3-padding">
1271                        <div class="w3-container"  style="margin-top:40px">
1272                            <h4 class="w3-center">Source Separation 2</h4>
1273                            <hr>
1274                            <a class="w3-text" href="taherian24_interspeech.html">
1275                                <p>
1276                                    Towards Explainable Monaural Speaker Separation with Auditory-based Training
1277                                    <br>
1278                                    <span class="w3-text w3-text-theme">
1279                                        Hassan Taherian, Vahid Ahmadi Kalkhorani, Ashutosh Pandey, Daniel Wong, Buye Xu, DeLiang Wang
1280                                    </span>
1281                                </p>
1282                            </a>
1283                            <a class="w3-text" href="ewert24_interspeech.html">
1284                                <p>
1285                                    Does the Lombard Effect Matter in Speech Separation? Introducing the Lombard-GRID-2mix Dataset
1286                                    <br>
1287                                    <span class="w3-text w3-text-theme">
1288                                        Iva Ewert, Marvin Borsdorf, Haizhou Li, Tanja Schultz
1289                                    </span>
1290                                </p>
1291                            </a>
1292                            <a class="w3-text" href="pan24_interspeech.html">
1293                                <p>
1294                                    PARIS: Pseudo-AutoRegressIve Siamese Training for Online Speech Separation
1295                                    <br>
1296                                    <span class="w3-text w3-text-theme">
1297                                        Zexu Pan, Gordon Wichern, François G. Germain, Kohei Saijo, Jonathan Le Roux
1298                                    </span>
1299                                </p>
1300                            </a>
1301                            <a class="w3-text" href="zhang24p_interspeech.html">
1302                                <p>
1303                                    OR-TSE: An Overlap-Robust Speaker Encoder for Target Speech Extraction
1304                                    <br>
1305                                    <span class="w3-text w3-text-theme">
1306                                        Yiru Zhang, Linyu Yao, Qun Yang
1307                                    </span>
1308                                </p>
1309                            </a>
1310                            <a class="w3-text" href="hsieh24b_interspeech.html">
1311                                <p>
1312                                    Multimodal Representation Loss Between Timed Text and Audio for Regularized Speech Separation
1313                                    <br>
1314                                    <span class="w3-text w3-text-theme">
1315                                        Tsun-An Hsieh, Heeyoul Choi, Minje Kim
1316                                    </span>
1317                                </p>
1318                            </a>
1319                            <a class="w3-text" href="lin24g_interspeech.html">
1320                                <p>
1321                                    SA-WavLM: Speaker-Aware Self-Supervised Pre-training for Mixture Speech
1322                                    <br>
1323                                    <span class="w3-text w3-text-theme">
1324                                        Jingru Lin, Meng Ge, Junyi Ao, Liqun Deng, Haizhou Li
1325                                    </span>
1326                                </p>
1327                            </a>
1328                            <a class="w3-text" href="wang24g_interspeech.html">
1329                                <p>
1330                                    TSE-PI: Target Sound Extraction under Reverberant Environments with Pitch Information
1331                                    <br>
1332                                    <span class="w3-text w3-text-theme">
1333                                        Yiwen Wang, Xihong Wu
1334                                    </span>
1335                                </p>
1336                            </a>
1337                            <a class="w3-text" href="saijo24_interspeech.html">
1338                                <p>
1339                                    Enhanced Reverberation as Supervision for Unsupervised Speech Separation
1340                                    <br>
1341                                    <span class="w3-text w3-text-theme">
1342                                        Kohei Saijo, Gordon Wichern, François G. Germain, Zexu Pan, Jonathan Le Roux
1343                                    </span>
1344                                </p>
1345                            </a>
1346                        </div>
1347                    </div>
1348                    <br>
1349                    <div class="w3-content" style="height:10px"  id="Noise Reduction, Dereverberation, and Echo Cancellation"></div>
1350                    <div class="w3-card w3-round w3-white w3-padding">
1351                        <div class="w3-container"  style="margin-top:40px">
1352                            <h4 class="w3-center">Noise Reduction, Dereverberation, and Echo Cancellation</h4>
1353                            <hr>
1354                            <a class="w3-text" href="zhao24_interspeech.html">
1355                                <p>
1356                                    Deep Echo Path Modeling for Acoustic Echo Cancellation
1357                                    <br>
1358                                    <span class="w3-text w3-text-theme">
1359                                        Fei Zhao, Chenggang Zhang, Shulin He, Jinjiang Liu, Xueliang Zhang
1360                                    </span>
1361                                </p>
1362                            </a>
1363                            <a class="w3-text" href="guo24_interspeech.html">
1364                                <p>
1365                                    Graph Attention Based Multi-Channel U-Net for Speech Dereverberation With Ad-Hoc Microphone Arrays
1366                                    <br>
1367                                    <span class="w3-text w3-text-theme">
1368                                        Hongmei Guo, Yijiang Chen, Xiao-Lei Zhang, Xuelong Li
1369                                    </span>
1370                                </p>
1371                            </a>
1372                            <a class="w3-text" href="bahrman24_interspeech.html">
1373                                <p>
1374                                    Speech dereverberation constrained on room impulse response characteristics
1375                                    <br>
1376                                    <span class="w3-text w3-text-theme">
1377                                        Louis Bahrman, Mathieu Fontaine, Jonathan Le Roux, Gaël Richard
1378                                    </span>
1379                                </p>
1380                            </a>
1381                            <a class="w3-text" href="yuan24_interspeech.html">
1382                                <p>
1383                                    DeWinder: Single-Channel Wind Noise Reduction using Ultrasound Sensing
1384                                    <br>
1385                                    <span class="w3-text w3-text-theme">
1386                                        Kuang Yuan, Shuo Han, Swarun Kumar, Bhiksha Raj
1387                                    </span>
1388                                </p>
1389                            </a>
1390                            <a class="w3-text" href="barnhill24_interspeech.html">
1391                                <p>
1392                                    ANIMAL-CLEAN – A Deep Denoising Toolkit for Animal-Independent Signal Enhancement
1393                                    <br>
1394                                    <span class="w3-text w3-text-theme">
1395                                        Alexander Barnhill, Elmar Noeth, Andreas Maier, Christian Bergler
1396                                    </span>
1397                                </p>
1398                            </a>
1399                            <a class="w3-text" href="nayak24b_interspeech.html">
1400                                <p>
1401                                    Elucidating Clock-drift Using Real-world Audios In Wireless Mode For Time-offset Insensitive End-to-End Asynchronous Acoustic Echo Cancellation
1402                                    <br>
1403                                    <span class="w3-text w3-text-theme">
1404                                        Premanand Nayak, M. Ali Basha Shaik
1405                                    </span>
1406                                </p>
1407                            </a>
1408                            <a class="w3-text" href="wang24o_interspeech.html">
1409                                <p>
1410                                    QMixCAT: Unsupervised Speech Enhancement Using Quality-guided Signal Mixing and Competitive Alternating Model Training
1411                                    <br>
1412                                    <span class="w3-text w3-text-theme">
1413                                        Shilin Wang, Haixin Guan, Yanhua Long
1414                                    </span>
1415                                </p>
1416                            </a>
1417                        </div>
1418                    </div>
1419                    <br>
1420                    <div class="w3-content" style="height:10px"  id="Computationally-Efficient Speech Enhancement"></div>
1421                    <div class="w3-card w3-round w3-white w3-padding">
1422                        <div class="w3-container"  style="margin-top:40px">
1423                            <h4 class="w3-center">Computationally-Efficient Speech Enhancement</h4>
1424                            <hr>
1425                            <a class="w3-text" href="bae24_interspeech.html">
1426                                <p>
1427                                    Speech Boosting: Low-Latency Live Speech Enhancement for TWS Earbuds
1428                                    <br>
1429                                    <span class="w3-text w3-text-theme">
1430                                        Hanbin Bae, Pavel Andreev, Azat Saginbaev, Nicholas Babaev, WonJun Lee, Hosang Sung, Hoon-Young Cho
1431                                    </span>
1432                                </p>
1433                            </a>
1434                            <a class="w3-text" href="gholami24_interspeech.html">
1435                                <p>
1436                                    Knowledge Distillation for Tiny Speech Enhancement with Latent Feature Augmentation
1437                                    <br>
1438                                    <span class="w3-text w3-text-theme">
1439                                        Behnam Gholami, Mostafa El-Khamy, KeeBong Song
1440                                    </span>
1441                                </p>
1442                            </a>
1443                            <a class="w3-text" href="zhang24o_interspeech.html">
1444                                <p>
1445                                    Sub-PNWR: Speech Enhancement Based on Signal Sub-Band Splitting and Pseudo Noisy Waveform Reconstruction Loss
1446                                    <br>
1447                                    <span class="w3-text w3-text-theme">
1448                                        Yuewei Zhang, Huanbin Zou, Jie Zhu
1449                                    </span>
1450                                </p>
1451                            </a>
1452                            <a class="w3-text" href="zhao24c_interspeech.html">
1453                                <p>
1454                                    Streamlining Speech Enhancement DNNs: an Automated Pruning Method Based on Dependency Graph with Advanced Regularized Loss Strategies
1455                                    <br>
1456                                    <span class="w3-text w3-text-theme">
1457                                        Zugang Zhao, Jinghong Zhang, Yonghui Liu, Jianbing Liu, Kai Niu, Zhiqiang He
1458                                    </span>
1459                                </p>
1460                            </a>
1461                            <a class="w3-text" href="zhang24k_interspeech.html">
1462                                <p>
1463                                    Lightweight Dynamic Sparse Transformer for Monaural Speech Enhancement
1464                                    <br>
1465                                    <span class="w3-text w3-text-theme">
1466                                        Zehua Zhang, Xuyi Zhuang, Yukun Qian, Mingjiang Wang
1467                                    </span>
1468                                </p>
1469                            </a>
1470                            <a class="w3-text" href="lin24h_interspeech.html">
1471                                <p>
1472                                    MUSE: Flexible Voiceprint Receptive Fields and Multi-Path Fusion Enhanced Taylor Transformer for U-Net-based Speech Enhancement
1473                                    <br>
1474                                    <span class="w3-text w3-text-theme">
1475                                        Zizhen Lin, Xiaoting Chen, Junyu Wang
1476                                    </span>
1477                                </p>
1478                            </a>
1479                            <a class="w3-text" href="cheng24_interspeech.html">
1480                                <p>
1481                                    Dynamic Gated Recurrent Neural Network for Compute-efficient Speech Enhancement
1482                                    <br>
1483                                    <span class="w3-text w3-text-theme">
1484                                        Longbiao Cheng, Ashutosh Pandey, Buye Xu, Tobi Delbruck, Shih-Chii Liu
1485                                    </span>
1486                                </p>
1487                            </a>
1488                        </div>
1489                    </div>
1490                    <br>
1491                    <div class="w3-content" style="height:10px"  id="Zero-shot TTS"></div>
1492                    <div class="w3-card w3-round w3-white w3-padding">
1493                        <div class="w3-container"  style="margin-top:40px">
1494                            <h4 class="w3-center">Zero-shot TTS</h4>
1495                            <hr>
1496                            <a class="w3-text" href="xue24c_interspeech.html">
1497                                <p>
1498                                    Improving Audio Codec-based Zero-Shot Text-to-Speech Synthesis with Multi-Modal Context and Large Language Model
1499                                    <br>
1500                                    <span class="w3-text w3-text-theme">
1501                                        Jinlong Xue, Yayue Deng, Yicheng Han, Yingming Gao, Ya Li
1502                                    </span>
1503                                </p>
1504                            </a>
1505                            <a class="w3-text" href="wang24v_interspeech.html">
1506                                <p>
1507                                    An Investigation of Noise Robustness for Flow-Matching-Based Zero-Shot TTS
1508                                    <br>
1509                                    <span class="w3-text w3-text-theme">
1510                                        Xiaofei Wang, Sefik Emre Eskimez, Manthan Thakker, Hemin Yang, Zirun Zhu, Min Tang, Yufei Xia, Jinzhu Li, Sheng Zhao, Jinyu Li, Naoyuki Kanda
1511                                    </span>
1512                                </p>
1513                            </a>
1514                            <a class="w3-text" href="fujita24b_interspeech.html">
1515                                <p>
1516                                    Lightweight Zero-shot Text-to-Speech with Mixture of Adapters
1517                                    <br>
1518                                    <span class="w3-text w3-text-theme">
1519                                        Kenichi Fujita, Takanori Ashihara, Marc Delcroix, Yusuke Ijima
1520                                    </span>
1521                                </p>
1522                            </a>
1523                            <a class="w3-text" href="pankov24_interspeech.html">
1524                                <p>
1525                                    DINO-VITS: Data-Efficient Zero-Shot TTS with Self-Supervised Speaker Verification Loss for Noise Robustness
1526                                    <br>
1527                                    <span class="w3-text w3-text-theme">
1528                                        Vikentii Pankov, Valeria Pronina, Alexander Kuzmin, Maksim Borisov, Nikita Usoltsev, Xingshan Zeng, Alexander Golubkov, Nikolai Ermolenko, Aleksandra Shirshova, Yulia Matveeva
1529                                    </span>
1530                                </p>
1531                            </a>
1532                        </div>
1533                    </div>
1534                    <br>
1535                    <div class="w3-content" style="height:10px"  id="Noise Robustness, Far-Field, and Multi-Talker ASR"></div>
1536                    <div class="w3-card w3-round w3-white w3-padding">
1537                        <div class="w3-container"  style="margin-top:40px">
1538                            <h4 class="w3-center">Noise Robustness, Far-Field, and Multi-Talker ASR</h4>
1539                            <hr>
1540                            <a class="w3-text" href="jin24_interspeech.html">
1541                                <p>
1542                                    LibriheavyMix: A 20,000-Hour Dataset for Single-Channel Reverberant Multi-Talker Speech Separation, ASR and Speaker Diarization
1543                                    <br>
1544                                    <span class="w3-text w3-text-theme">
1545                                        Zengrui Jin, Yifan Yang, Mohan Shi, Wei Kang, Xiaoyu Yang, Zengwei Yao, Fangjun Kuang, Liyong Guo, Lingwei Meng, Long Lin, Yong Xu, Shi-Xiong Zhang, Daniel Povey
1546                                    </span>
1547                                </p>
1548                            </a>
1549                            <a class="w3-text" href="xing24_interspeech.html">
1550                                <p>
1551                                    A Joint Noise Disentanglement and Adversarial Training Framework for Robust Speaker Verification
1552                                    <br>
1553                                    <span class="w3-text w3-text-theme">
1554                                        Xujiang Xing, Mingxing Xu, Thomas Fang Zheng
1555                                    </span>
1556                                </p>
1557                            </a>
1558                            <a class="w3-text" href="shi24e_interspeech.html">
1559                                <p>
1560                                    Serialized Output Training by Learned Dominance
1561                                    <br>
1562                                    <span class="w3-text w3-text-theme">
1563                                        Ying Shi, Lantian Li, Shi Yin, Dong Wang, Jiqing Han
1564                                    </span>
1565                                </p>
1566                            </a>
1567                            <a class="w3-text" href="zheng24d_interspeech.html">
1568                                <p>
1569                                    SOT Triggered Neural Clustering for Speaker Attributed ASR
1570                                    <br>
1571                                    <span class="w3-text w3-text-theme">
1572                                        Xianrui Zheng, Guangzhi Sun, Chao Zhang, Philip C. Woodland
1573                                    </span>
1574                                </p>
1575                            </a>
1576                            <a class="w3-text" href="bando24_interspeech.html">
1577                                <p>
1578                                    Neural Blind Source Separation and Diarization for Distant Speech Recognition
1579                                    <br>
1580                                    <span class="w3-text w3-text-theme">
1581                                        Yoshiaki Bando, Tomohiko Nakamura, Shinji Watanabe
1582                                    </span>
1583                                </p>
1584                            </a>
1585                            <a class="w3-text" href="masumura24_interspeech.html">
1586                                <p>
1587                                    Unified Multi-Talker ASR with and without Target-speaker Enrollment
1588                                    <br>
1589                                    <span class="w3-text w3-text-theme">
1590                                        Ryo Masumura, Naoki Makishima, Tomohiro Tanaka, Mana Ihori, Naotaka Kawata, Shota Orihashi, Kazutoshi Shinoda, Taiga Yamane, Saki Mizuno, Keita Suzuki, Satoshi Suzuki, Nobukatsu Hojo, Takafumi Moriya, Atsushi Ando
1591                                    </span>
1592                                </p>
1593                            </a>
1594                        </div>
1595                    </div>
1596                    <br>
1597                    <div class="w3-content" style="height:10px"  id="Contextual Biasing and Adaptation"></div>
1598                    <div class="w3-card w3-round w3-white w3-padding">
1599                        <div class="w3-container"  style="margin-top:40px">
1600                            <h4 class="w3-center">Contextual Biasing and Adaptation</h4>
1601                            <hr>
1602                            <a class="w3-text" href="shamsian24_interspeech.html">
1603                                <p>
1604                                    Keyword-Guided Adaptation of Automatic Speech Recognition
1605                                    <br>
1606                                    <span class="w3-text w3-text-theme">
1607                                        Aviv Shamsian, Aviv Navon, Neta Glazer, Gill Hetz, Joseph Keshet
1608                                    </span>
1609                                </p>
1610                            </a>
1611                            <a class="w3-text" href="manhtienanh24_interspeech.html">
1612                                <p>
1613                                    Improving Speech Recognition with Prompt-based Contextualized ASR and LLM-based Re-predictor
1614                                    <br>
1615                                    <span class="w3-text w3-text-theme">
1616                                        Nguyen Manh Tien Anh, Thach Ho Sy
1617                                    </span>
1618                                </p>
1619                            </a>
1620                            <a class="w3-text" href="wang24q_interspeech.html">
1621                                <p>
1622                                    Incorporating Class-based Language Model for Named Entity Recognition in Factorized Neural Transducer
1623                                    <br>
1624                                    <span class="w3-text w3-text-theme">
1625                                        Peng Wang, Yifan Yang, Zheng Liang, Tian Tan, Shiliang Zhang, Xie Chen
1626                                    </span>
1627                                </p>
1628                            </a>
1629                            <a class="w3-text" href="yang24j_interspeech.html">
1630                                <p>
1631                                    Contextual Biasing with Confidence-based Homophone Detector for Mandarin End-to-End Speech Recognition
1632                                    <br>
1633                                    <span class="w3-text w3-text-theme">
1634                                        Chengxu Yang, Lin Zheng, Sanli Tian, Gaofeng Cheng, Sujie Xiao, Ta Li
1635                                    </span>
1636                                </p>
1637                            </a>
1638                            <a class="w3-text" href="huang24f_interspeech.html">
1639                                <p>
1640                                    Improving Neural Biasing for Contextual Speech Recognition by Early Context Injection and Text Perturbation
1641                                    <br>
1642                                    <span class="w3-text w3-text-theme">
1643                                        Ruizhe Huang, Mahsa Yarmohammadi, Sanjeev Khudanpur, Daniel Povey
1644                                    </span>
1645                                </p>
1646                            </a>
1647                            <a class="w3-text" href="andrusenko24_interspeech.html">
1648                                <p>
1649                                    Fast Context-Biasing for CTC and Transducer ASR models with CTC-based Word Spotter
1650                                    <br>
1651                                    <span class="w3-text w3-text-theme">
1652                                        Andrei Andrusenko, Aleksandr Laptev, Vladimir Bataev, Vitaly Lavrukhin, Boris Ginsburg
1653                                    </span>
1654                                </p>
1655                            </a>
1656                            <a class="w3-text" href="wei24_interspeech.html">
1657                                <p>
1658                                    Prompt Tuning for Speech Recognition on Unknown Spoken Name Entities
1659                                    <br>
1660                                    <span class="w3-text w3-text-theme">
1661                                        Xizi Wei, Stephen McGregor
1662                                    </span>
1663                                </p>
1664                            </a>
1665                            <a class="w3-text" href="liu24e_interspeech.html">
1666                                <p>
1667                                    Improved Factorized Neural Transducer Model For Text-only Domain Adaptation
1668                                    <br>
1669                                    <span class="w3-text w3-text-theme">
1670                                        Junzhe Liu, Jianwei Yu, Xie Chen
1671                                    </span>
1672                                </p>
1673                            </a>
1674                            <a class="w3-text" href="liu24d_interspeech.html">
1675                                <p>
1676                                    Modality Translation Learning for Joint Speech-Text Model
1677                                    <br>
1678                                    <span class="w3-text w3-text-theme">
1679                                        Pin-Yen Liu, Jen-Tzung Chien
1680                                    </span>
1681                                </p>
1682                            </a>
1683                            <a class="w3-text" href="zhao24d_interspeech.html">
1684                                <p>
1685                                    SAML: Speaker Adaptive Mixture of LoRA Experts for End-to-End ASR
1686                                    <br>
1687                                    <span class="w3-text w3-text-theme">
1688                                        Qiuming Zhao, Guangzhi Sun, Chao Zhang, Mingxing Xu, Thomas Fang Zheng
1689                                    </span>
1690                                </p>
1691                            </a>
1692                            <a class="w3-text" href="ando24_interspeech.html">
1693                                <p>
1694                                    Factor-Conditioned Speaking-Style Captioning
1695                                    <br>
1696                                    <span class="w3-text w3-text-theme">
1697                                        Atsushi Ando, Takafumi Moriya, Shota Horiguchi, Ryo Masumura
1698                                    </span>
1699                                </p>
1700                            </a>
1701                            <a class="w3-text" href="khassanov24_interspeech.html">
1702                                <p>
1703                                    Dual-Pipeline with Low-Rank Adaptation for New Language Integration in Multilingual ASR
1704                                    <br>
1705                                    <span class="w3-text w3-text-theme">
1706                                        Yerbolat Khassanov, Zhipeng Chen, Tianfeng Chen, Tze Yuang Chong, Wei Li, Jun Zhang, Lu Lu, Yuxuan Wang
1707                                    </span>
1708                                </p>
1709                            </a>
1710                            <a class="w3-text" href="yusuf24_interspeech.html">
1711                                <p>
1712                                    Speculative Speech Recognition by Audio-Prefixed Low-Rank Adaptation of Language Models
1713                                    <br>
1714                                    <span class="w3-text w3-text-theme">
1715                                        Bolaji Yusuf, Murali Karthick Baskar, Andrew Rosenberg, Bhuvana Ramabhadran
1716                                    </span>
1717                                </p>
1718                            </a>
1719                            <a class="w3-text" href="kim24u_interspeech.html">
1720                                <p>
1721                                    Domain-Aware Data Selection for Speech Classification via Meta-Reweighting
1722                                    <br>
1723                                    <span class="w3-text w3-text-theme">
1724                                        Junghun Kim, Ka Hyun Park, Hoyoung Yoon, U Kang
1725                                    </span>
1726                                </p>
1727                            </a>
1728                        </div>
1729                    </div>
1730                    <br>
1731                    <div class="w3-content" style="height:10px"  id="Spoken Language Understanding"></div>
1732                    <div class="w3-card w3-round w3-white w3-padding">
1733                        <div class="w3-container"  style="margin-top:40px">
1734                            <h4 class="w3-center">Spoken Language Understanding</h4>
1735                            <hr>
1736                            <a class="w3-text" href="futami24_interspeech.html">
1737                                <p>
1738                                    Finding Task-specific Subnetworks in Multi-task Spoken Language Understanding Model
1739                                    <br>
1740                                    <span class="w3-text w3-text-theme">
1741                                        Hayato Futami, Siddhant Arora, Yosuke Kashiwagi, Emiru Tsunoo, Shinji Watanabe
1742                                    </span>
1743                                </p>
1744                            </a>
1745                            <a class="w3-text" href="porjazovski24_interspeech.html">
1746                                <p>
1747                                    Out-of-distribution generalisation in spoken language understanding
1748                                    <br>
1749                                    <span class="w3-text w3-text-theme">
1750                                        Dejan Porjazovski, Anssi Moisio, Mikko Kurimo
1751                                    </span>
1752                                </p>
1753                            </a>
1754                            <a class="w3-text" href="laperriere24_interspeech.html">
1755                                <p>
1756                                    A dual task learning approach to fine-tune a multilingual semantic speech encoder for Spoken Language Understanding
1757                                    <br>
1758                                    <span class="w3-text w3-text-theme">
1759                                        Gaëlle Laperrière, Sahar Ghannay, Bassam Jabaian, Yannick Estève
1760                                    </span>
1761                                </p>
1762                            </a>
1763                            <a class="w3-text" href="lee24i_interspeech.html">
1764                                <p>
1765                                    Speech-MASSIVE: A Multilingual Speech Dataset for SLU and Beyond
1766                                    <br>
1767                                    <span class="w3-text w3-text-theme">
1768                                        Beomseok Lee, Ioan Calapodescu, Marco Gaido, Matteo Negri, Laurent Besacier
1769                                    </span>
1770                                </p>
1771                            </a>
1772                            <a class="w3-text" href="li24b_interspeech.html">
1773                                <p>
1774                                    Using Large Language Model for End-to-End Chinese ASR and NER
1775                                    <br>
1776                                    <span class="w3-text w3-text-theme">
1777                                        Yuang Li, Jiawei Yu, Min Zhang, Mengxin Ren, Yanqing Zhao, Xiaofeng Zhao, Shimin Tao, Jinsong Su, Hao Yang
1778                                    </span>
1779                                </p>
1780                            </a>
1781                            <a class="w3-text" href="koudounas24b_interspeech.html">
1782                                <p>
1783                                    A Contrastive Learning Approach to Mitigate Bias in Speech Models
1784                                    <br>
1785                                    <span class="w3-text w3-text-theme">
1786                                        Alkis Koudounas, Flavio Giobergia, Eliana Pastor, Elena Baralis
1787                                    </span>
1788                                </p>
1789                            </a>
1790                        </div>
1791                    </div>
1792                    <br>
1793                    <div class="w3-content" style="height:10px"  id="Spoken Machine Translation 1"></div>
1794                    <div class="w3-card w3-round w3-white w3-padding">
1795                        <div class="w3-container"  style="margin-top:40px">
1796                            <h4 class="w3-center">Spoken Machine Translation 1</h4>
1797                            <hr>
1798                            <a class="w3-text" href="huang24h_interspeech.html">
1799                                <p>
1800                                    Investigating Decoder-only Large Language Models for Speech-to-text Translation
1801                                    <br>
1802                                    <span class="w3-text w3-text-theme">
1803                                        Chao-Wei Huang, Hui Lu, Hongyu Gong, Hirofumi Inaguma, Ilia Kulikov, Ruslan Mavlyutov, Sravya Popuri
1804                                    </span>
1805                                </p>
1806                            </a>
1807                            <a class="w3-text" href="hirschkind24_interspeech.html">
1808                                <p>
1809                                    Diffusion Synthesizer for Efficient Multilingual Speech to Speech Translation
1810                                    <br>
1811                                    <span class="w3-text w3-text-theme">
1812                                        Nameer Hirschkind, Xiao Yu, Mahesh Kumar Nandwana, Joseph Liu, Eloi DuBois, Dao Le, Nicolas Thiebaut, Colin Sinclair, Kyle Spence, Charles Shang, Zoe Abrams, Morgan McGuire
1813                                    </span>
1814                                </p>
1815                            </a>
1816                            <a class="w3-text" href="chen24v_interspeech.html">
1817                                <p>
1818                                    Sign Value Constraint Decomposition for Efficient 1-Bit Quantization of Speech Translation Tasks
1819                                    <br>
1820                                    <span class="w3-text w3-text-theme">
1821                                        Nan Chen, Yonghe Wang, Feilong Bao
1822                                    </span>
1823                                </p>
1824                            </a>
1825                            <a class="w3-text" href="lee24h_interspeech.html">
1826                                <p>
1827                                    Lightweight Audio Segmentation for Long-form Speech Translation
1828                                    <br>
1829                                    <span class="w3-text w3-text-theme">
1830                                        Jaesong Lee, Soyoon Kim, Hanbyul Kim, Joon Son Chung
1831                                    </span>
1832                                </p>
1833                            </a>
1834                            <a class="w3-text" href="tan24b_interspeech.html">
1835                                <p>
1836                                    Contrastive Feedback Mechanism for Simultaneous Speech Translation
1837                                    <br>
1838                                    <span class="w3-text w3-text-theme">
1839                                        Haotian Tan, Sakriani Sakti
1840                                    </span>
1841                                </p>
1842                            </a>
1843                            <a class="w3-text" href="macaire24_interspeech.html">
1844                                <p>
1845                                    Towards Speech-to-Pictograms Translation
1846                                    <br>
1847                                    <span class="w3-text w3-text-theme">
1848                                        Cécile Macaire, Chloé Dion, Didier Schwab, Benjamin Lecouteux, Emmanuelle Esperança-Rodier
1849                                    </span>
1850                                </p>
1851                            </a>
1852                        </div>
1853                    </div>
1854                    <br>
1855                    <div class="w3-content" style="height:10px"  id="Hearing Disorders"></div>
1856                    <div class="w3-card w3-round w3-white w3-padding">
1857                        <div class="w3-container"  style="margin-top:40px">
1858                            <h4 class="w3-center">Hearing Disorders</h4>
1859                            <hr>
1860                            <a class="w3-text" href="lee24e_interspeech.html">
1861                                <p>
1862                                    Automatic Assessment of Speech Production Skills for Children with Cochlear Implants Using Wav2Vec2.0 Acoustic Embeddings
1863                                    <br>
1864                                    <span class="w3-text w3-text-theme">
1865                                        Seonwoo Lee, Sunhee Kim, Minhwa Chung
1866                                    </span>
1867                                </p>
1868                            </a>
1869                            <a class="w3-text" href="ahn24_interspeech.html">
1870                                <p>
1871                                    SyncVSR: Data-Efficient Visual Speech Recognition with End-to-End Crossmodal Audio Token Synchronization
1872                                    <br>
1873                                    <span class="w3-text w3-text-theme">
1874                                        Young Jin Ahn, Jungwoo Park, Sangha Park, Jonghyun Choi, Kee-Eung Kim
1875                                    </span>
1876                                </p>
1877                            </a>
1878                            <a class="w3-text" href="huckvale24_interspeech.html">
1879                                <p>
1880                                    Evaluating a 3-factor listener model for prediction of speech intelligibility to hearing-impaired listeners
1881                                    <br>
1882                                    <span class="w3-text w3-text-theme">
1883                                        Mark Huckvale, Gaston Hilkhuysen
1884                                    </span>
1885                                </p>
1886                            </a>
1887                            <a class="w3-text" href="fagniart24_interspeech.html">
1888                                <p>
1889                                    Production of fricative consonants in French-speaking children with cochlear implants and typical hearing: acoustic and phonological analyses.
1890                                    <br>
1891                                    <span class="w3-text w3-text-theme">
1892                                        Sophie Fagniart, Brigitte Charlier, Véronique Delvaux, Bernard Harmegnies, Anne Huberlant, Myriam Piccaluga, Kathy Huet
1893                                    </span>
1894                                </p>
1895                            </a>
1896                            <a class="w3-text" href="irino24_interspeech.html">
1897                                <p>
1898                                    Signal processing algorithm effective for sound quality of hearing loss simulators 
1899                                    <br>
1900                                    <span class="w3-text w3-text-theme">
1901                                        Toshio Irino, Shintaro Doan, Minami Ishikawa
1902                                    </span>
1903                                </p>
1904                            </a>
1905                            <a class="w3-text" href="niu24c_interspeech.html">
1906                                <p>
1907                                    Auditory Spatial Attention Detection Based on Feature Disentanglement and Brain Connectivity-Informed Graph Neural Networks
1908                                    <br>
1909                                    <span class="w3-text w3-text-theme">
1910                                        Yixiang Niu, Ning Chen, Hongqing Zhu, Zhiying Zhu, Guangqiang Li, Yibo Chen
1911                                    </span>
1912                                </p>
1913                            </a>
1914                            <a class="w3-text" href="monaghan24_interspeech.html">
1915                                <p>
1916                                    Automatic Detection of Hearing Loss from Children's Speech using wav2vec 2.0 Features
1917                                    <br>
1918                                    <span class="w3-text w3-text-theme">
1919                                        Jessica Monaghan, Arun Sebastian, Nicky Chong-White, Vicky Zhang, Vijayalakshmi Easwar, Padraig Kitterick
1920                                    </span>
1921                                </p>
1922                            </a>
1923                        </div>
1924                    </div>
1925                    <br>
1926                    <div class="w3-content" style="height:10px"  id="Speech Disorders 2"></div>
1927                    <div class="w3-card w3-round w3-white w3-padding">
1928                        <div class="w3-container"  style="margin-top:40px">
1929                            <h4 class="w3-center">Speech Disorders 2</h4>
1930                            <hr>
1931                            <a class="w3-text" href="changawala24_interspeech.html">
1932                                <p>
1933                                    Whister: Using Whisper’s representations for Stuttering detection
1934                                    <br>
1935                                    <span class="w3-text w3-text-theme">
1936                                        Vrushank Changawala, Frank Rudzicz
1937                                    </span>
1938                                </p>
1939                            </a>
1940                            <a class="w3-text" href="xiong24_interspeech.html">
1941                                <p>
1942                                    Improving Speech-Based Dysarthria Detection using Multi-task Learning with Gradient Projection
1943                                    <br>
1944                                    <span class="w3-text w3-text-theme">
1945                                        Yan Xiong, Visar Berisha, Julie Liss, Chaitali Chakrabarti
1946                                    </span>
1947                                </p>
1948                            </a>
1949                            <a class="w3-text" href="chen24i_interspeech.html">
1950                                <p>
1951                                    Cascaded Transfer Learning Strategy for Cross-Domain Alzheimer's  Disease Recognition through Spontaneous Speech
1952                                    <br>
1953                                    <span class="w3-text w3-text-theme">
1954                                        Guanlin Chen, Yun Jin
1955                                    </span>
1956                                </p>
1957                            </a>
1958                            <a class="w3-text" href="ilias24_interspeech.html">
1959                                <p>
1960                                    A Cross-Attention Layer coupled with Multimodal Fusion Methods for Recognizing Depression from Spontaneous Speech
1961                                    <br>
1962                                    <span class="w3-text w3-text-theme">
1963                                        Loukas Ilias, Dimitris Askounis
1964                                    </span>
1965                                </p>
1966                            </a>
1967                            <a class="w3-text" href="ng24_interspeech.html">
1968                                <p>
1969                                    Segmental and Suprasegmental Speech Foundation Models for Classifying Cognitive Risk Factors: Evaluating Out-of-the-Box Performance
1970                                    <br>
1971                                    <span class="w3-text w3-text-theme">
1972                                        Si-Ioi Ng, Lingfeng Xu, Kimberly D. Mueller, Julie Liss, Visar Berisha
1973                                    </span>
1974                                </p>
1975                            </a>
1976                            <a class="w3-text" href="papadimitriou24_interspeech.html">
1977                                <p>
1978                                    Multimodal Continuous Fingerspelling Recognition via Visual Alignment Learning
1979                                    <br>
1980                                    <span class="w3-text w3-text-theme">
1981                                        Katerina Papadimitriou, Gerasimos Potamianos
1982                                    </span>
1983                                </p>
1984                            </a>
1985                            <a class="w3-text" href="ariasvergara24_interspeech.html">
1986                                <p>
1987                                    Contrastive Learning Approach for Assessment of Phonological Precision in Patients with Tongue Cancer Using MRI Data
1988                                    <br>
1989                                    <span class="w3-text w3-text-theme">
1990                                        Tomas Arias-Vergara, Paula Andrea Pérez-Toro, Xiaofeng Liu, Fangxu Xing, Maureen Stone, Jiachen Zhuo, Jerry L. Prince, Maria Schuster, Elmar Noeth, Jonghye Woo, Andreas Maier
1991                                    </span>
1992                                </p>
1993                            </a>
1994                            <a class="w3-text" href="zhang24l_interspeech.html">
1995                                <p>
1996                                    DysArinVox: DYSphonia &amp; DYSarthria mandARIN speech corpus
1997                                    <br>
1998                                    <span class="w3-text w3-text-theme">
1999                                        Haojie Zhang, Tao Zhang, Ganjun Liu, Dehui Fu, Xiaohui Hou, Ying Lv
2000                                    </span>
2001                                </p>
2002                            </a>
2003                            <a class="w3-text" href="zhou24e_interspeech.html">
2004                                <p>
2005                                    YOLO-Stutter: End-to-end Region-Wise Speech Dysfluency Detection
2006                                    <br>
2007                                    <span class="w3-text w3-text-theme">
2008                                        Xuanru Zhou, Anshul Kashyap, Steve Li, Ayati Sharma, Brittany Morin, David Baquirin, Jet Vonk, Zoe Ezzes, Zachary Miller, Maria Tempini, Jiachen Lian, Gopala Anumanchipalli
2009                                    </span>
2010                                </p>
2011                            </a>
2012                            <a class="w3-text" href="gosztolya24c_interspeech.html">
2013                                <p>
2014                                    Automatic Longitudinal Investigation of Multiple Sclerosis Subjects
2015                                    <br>
2016                                    <span class="w3-text w3-text-theme">
2017                                        Gábor Gosztolya, Veronika Svindt, Judit Bóna, Ildikó Hoffmann
2018                                    </span>
2019                                </p>
2020                            </a>
2021                        </div>
2022                    </div>
2023                    <br>
2024                    <div class="w3-content" style="height:10px"  id="TAUKADIAL Challenge: Speech-Based Cognitive Assessment in Chinese and English (Special Session)"></div>
2025                    <div class="w3-card w3-round w3-white w3-padding">
2026                        <div class="w3-container"  style="margin-top:40px">
2027                            <h4 class="w3-center">TAUKADIAL Challenge: Speech-Based Cognitive Assessment in Chinese and English (Special Session)</h4>
2028                            <hr>
2029                            <a class="w3-text" href="luz24_interspeech.html">
2030                                <p>
2031                                    Connected Speech-Based Cognitive Assessment in Chinese and English
2032                                    <br>
2033                                    <span class="w3-text w3-text-theme">
2034                                        Saturnino Luz, Sofia De La Fuente Garcia, Fasih Haider, Davida Fromm, Brian MacWhinney, Alyssa Lanzi, Ya-Ning Chang, Chia-Ju Chou, Yi-Chien Liu
2035                                    </span>
2036                                </p>
2037                            </a>
2038                            <a class="w3-text" href="ortizperez24_interspeech.html">
2039                                <p>
2040                                    Cognitive Insights Across Languages: Enhancing Multimodal Interview Analysis
2041                                    <br>
2042                                    <span class="w3-text w3-text-theme">
2043                                        David Ortiz-Perez, Jose Garcia-Rodriguez, David Tomás
2044                                    </span>
2045                                </p>
2046                            </a>
2047                            <a class="w3-text" href="gosztolya24_interspeech.html">
2048                                <p>
2049                                    Combining Acoustic Feature Sets for Detecting Mild Cognitive Impairment in the Interspeech'24 TAUKADIAL Challenge
2050                                    <br>
2051                                    <span class="w3-text w3-text-theme">
2052                                        Gábor Gosztolya, László Tóth
2053                                    </span>
2054                                </p>
2055                            </a>
2056                            <a class="w3-text" href="duan24_interspeech.html">
2057                                <p>
2058                                    Pre-trained Feature Fusion and Matching for Mild Cognitive Impairment Detection
2059                                    <br>
2060                                    <span class="w3-text w3-text-theme">
2061                                        Junwen Duan, Fangyuan Wei, Hong-Dong Li, Jin Liu
2062                                    </span>
2063                                </p>
2064                            </a>
2065                            <a class="w3-text" href="barreraaltuna24_interspeech.html">
2066                                <p>
2067                                    The Interspeech 2024 TAUKADIAL Challenge: Multilingual Mild Cognitive Impairment Detection with Multimodal Approach
2068                                    <br>
2069                                    <span class="w3-text w3-text-theme">
2070                                        Benjamin Barrera-Altuna, Daeun Lee, Zaima Zarnaz, Jinyoung Han, Seungbae Kim
2071                                    </span>
2072                                </p>
2073                            </a>
2074                            <a class="w3-text" href="favaro24_interspeech.html">
2075                                <p>
2076                                    Leveraging Universal Speech Representations for Detecting and Assessing the Severity of Mild Cognitive Impairment Across Languages
2077                                    <br>
2078                                    <span class="w3-text w3-text-theme">
2079                                        Anna Favaro, Tianyu Cao, Najim Dehak, Laureano Moro-Velazquez
2080                                    </span>
2081                                </p>
2082                            </a>
2083                            <a class="w3-text" href="hoang24_interspeech.html">
2084                                <p>
2085                                    Translingual Language Markers for Cognitive Assessment from Spontaneous Speech
2086                                    <br>
2087                                    <span class="w3-text w3-text-theme">
2088                                        Bao Hoang, Yijiang Pang, Hiroko Dodge, Jiayu Zhou
2089                                    </span>
2090                                </p>
2091                            </a>
2092                            <a class="w3-text" href="pereztoro24_interspeech.html">
2093                                <p>
2094                                    Multilingual Speech and Language Analysis for the Assessment of Mild Cognitive Impairment: Outcomes from the Taukadial Challenge
2095                                    <br>
2096                                    <span class="w3-text w3-text-theme">
2097                                        Paula Andrea Pérez-Toro, Tomas Arias-Vergara, Philipp Klumpp, Tobias Weise, Maria Schuster, Elmar Noeth, Juan Rafael Orozco-Arroyave, Andreas Maier
2098                                    </span>
2099                                </p>
2100                            </a>
2101                        </div>
2102                    </div>
2103                    <br>
2104                    <div class="w3-content" style="height:10px"  id="Show and Tell 1"></div>
2105                    <div class="w3-card w3-round w3-white w3-padding">
2106                        <div class="w3-container"  style="margin-top:40px">
2107                            <h4 class="w3-center">Show and Tell 1</h4>
2108                            <hr>
2109                            <a class="w3-text" href="arai24_interspeech.html">
2110                                <p>
2111                                    Production of phrases by mechanical models of the human vocal tract
2112                                    <br>
2113                                    <span class="w3-text w3-text-theme">
2114                                        Takayuki Arai, Ryohei Suzuki, Chandler Earp, Shinya Tsuji, Keiko Ochi
2115                                    </span>
2116                                </p>
2117                            </a>
2118                            <a class="w3-text" href="gourav24_interspeech.html">
2119                                <p>
2120                                    Faster Vocoder: a multi threading approach to achieve low latency during TTS Inference
2121                                    <br>
2122                                    <span class="w3-text w3-text-theme">
2123                                        Vishal Gourav, Ankit Tyagi, Phanindra Mankale
2124                                    </span>
2125                                </p>
2126                            </a>
2127                            <a class="w3-text" href="mohan24_interspeech.html">
2128                                <p>
2129                                    A powerful and modern AAC composition tool for impaired speakers
2130                                    <br>
2131                                    <span class="w3-text w3-text-theme">
2132                                        Aanchan Mohan, Monideep Chakraborti, Katelyn Eng, Nailia Kushaeva, Mirjana Prpa, Jordan Lewis, Tianyi Zhang, Vince Geisler, Carol Geisler
2133                                    </span>
2134                                </p>
2135                            </a>
2136                            <a class="w3-text" href="mika24_interspeech.html">
2137                                <p>
2138                                    VoxFlow AI: wearable voice converter for atypical speech
2139                                    <br>
2140                                    <span class="w3-text w3-text-theme">
2141                                        Grzegorz P. Mika, Konrad Zieli´nski, Paweł Cyrta, Marek Grzelec
2142                                    </span>
2143                                </p>
2144                            </a>
2145                            <a class="w3-text" href="akarsh24_interspeech.html">
2146                                <p>
2147                                    Stress transfer in speech-to-speech machine translation
2148                                    <br>
2149                                    <span class="w3-text w3-text-theme">
2150                                        Sai Akarsh, Vamshiraghusimha Narasinga, Anil Kumar Vuppala
2151                                    </span>
2152                                </p>
2153                            </a>
2154                            <a class="w3-text" href="okamoto24b_interspeech.html">
2155                                <p>
2156                                    Mobile PresenTra: NICT fast neural text-to-speech system on smartphones with incremental inference of MS-FC-HiFi-GAN for law-latency synthesis
2157                                    <br>
2158                                    <span class="w3-text w3-text-theme">
2159                                        Takuma Okamoto, Yamato Ohtani, Hisashi Kawai
2160                                    </span>
2161                                </p>
2162                            </a>
2163                            <a class="w3-text" href="peirolilja24_interspeech.html">
2164                                <p>
2165                                    Multi-speaker and multi-dialectal Catalan TTS models for video gaming
2166                                    <br>
2167                                    <span class="w3-text w3-text-theme">
2168                                        Alex Peiró-Lilja, José Giraldo, Martí Llopart-Font, Carme Armentano-Oller, Baybars Külebi, Mireia Farrús
2169                                    </span>
2170                                </p>
2171                            </a>
2172                            <a class="w3-text" href="francis24_interspeech.html">
2173                                <p>
2174                                    ConnecTone: a modular AAC system prototype with contextual generative text prediction and style-adaptive conversational TTS
2175                                    <br>
2176                                    <span class="w3-text w3-text-theme">
2177                                        Juliana Francis, Éva Székely, Joakim Gustafson
2178                                    </span>
2179                                </p>
2180                            </a>
2181                            <a class="w3-text" href="rohmatillah24_interspeech.html">
2182                                <p>
2183                                    Reliable dialogue system for facilitating student-counselor communication
2184                                    <br>
2185                                    <span class="w3-text w3-text-theme">
2186                                        Mahdin Rohmatillah, Bryan Gautama Ngo, Willianto Sulaiman, Po-Chuan Chen, Jen-Tzung Chien
2187                                    </span>
2188                                </p>
2189                            </a>
2190                            <a class="w3-text" href="lameris24_interspeech.html">
2191                                <p>
2192                                    CreakVC: a voice conversion tool for modulating creaky voice
2193                                    <br>
2194                                    <span class="w3-text w3-text-theme">
2195                                        Harm Lameris, Joakim Gustafson, Éva Székely
2196                                    </span>
2197                                </p>
2198                            </a>
2199                            <a class="w3-text" href="tsao24_interspeech.html">
2200                                <p>
2201                                    EZTalking: English assessment platform for teachers and students
2202                                    <br>
2203                                    <span class="w3-text w3-text-theme">
2204                                        Yu-Sheng Tsao, Yung-Chang Hsu, Jiun-Ting Li, Siang-Hong Weng, Tien-Hong Lo, Berlin Chen
2205                                    </span>
2206                                </p>
2207                            </a>
2208                        </div>
2209                    </div>
2210                    <br>
2211                    <div class="w3-content" style="height:10px"  id="Keynote 2"></div>
2212                    <div class="w3-card w3-round w3-white w3-padding">
2213                        <div class="w3-container"  style="margin-top:40px">
2214                            <h4 class="w3-center">Keynote 2</h4>
2215                            <hr>
2216                            <a class="w3-text" href="araki24_interspeech.html">
2217                                <p>
2218                                    Frontier of Frontend for Conversational Speech Processing
2219                                    <br>
2220                                    <span class="w3-text w3-text-theme">
2221                                        Shoko Araki
2222                                    </span>
2223                                </p>
2224                            </a>
2225                        </div>
2226                    </div>
2227                    <br>
2228                    <div class="w3-content" style="height:10px"  id="Phonetics and Phonology of Second Language Acquisition"></div>
2229                    <div class="w3-card w3-round w3-white w3-padding">
2230                        <div class="w3-container"  style="margin-top:40px">
2231                            <h4 class="w3-center">Phonetics and Phonology of Second Language Acquisition</h4>
2232                            <hr>
2233                            <a class="w3-text" href="tuttosi24_interspeech.html">
2234                                <p>
2235                                    Mmm whatcha say? Uncovering distal and proximal context effects in first and second-language word perception using psychophysical reverse correlation
2236                                    <br>
2237                                    <span class="w3-text w3-text-theme">
2238                                        Paige Tuttösí, H. Henny Yeung, Yue Wang, Fenqi Wang, Guillaume Denis, Jean-Julien Aucouturier, Angelica Lim
2239                                    </span>
2240                                </p>
2241                            </a>
2242                            <a class="w3-text" href="popescu24_interspeech.html">
2243                                <p>
2244                                    Automatic Speech Recognition with parallel L1 and L2 acoustic phone models to evaluate /l/ allophony in L2 English speech production
2245                                    <br>
2246                                    <span class="w3-text w3-text-theme">
2247                                        Anisia Popescu, Lori Lamel, Ioana Vasilescu, Laurence Devillers
2248                                    </span>
2249                                </p>
2250                            </a>
2251                            <a class="w3-text" href="huang24i_interspeech.html">
2252                                <p>
2253                                    Analysis of articulatory setting for L1 and L2 English speakers using MRI data
2254                                    <br>
2255                                    <span class="w3-text w3-text-theme">
2256                                        Kevin Huang, Jack Goldberg, Louis Goldstein, Shrikanth Narayanan
2257                                    </span>
2258                                </p>
2259                            </a>
2260                            <a class="w3-text" href="colgiu24_interspeech.html">
2261                                <p>
2262                                    Bilingual Rhotic Production Patterns: A Generational Comparison of Spanish-English Bilingual Speakers in Canada
2263                                    <br>
2264                                    <span class="w3-text w3-text-theme">
2265                                        Ioana Colgiu, Laura Spinu, Rajiv Rao, Yasaman Rafat
2266                                    </span>
2267                                </p>
2268                            </a>
2269                            <a class="w3-text" href="coulange24_interspeech.html">
2270                                <p>
2271                                    Exploring Impact of Pausing and Lexical Stress Patterns on L2 English Comprehensibility in Real Time
2272                                    <br>
2273                                    <span class="w3-text w3-text-theme">
2274                                        Sylvain Coulange, Tsuneo Kato, Solange Rossato, Monica Masperi
2275                                    </span>
2276                                </p>
2277                            </a>
2278                            <a class="w3-text" href="wu24m_interspeech.html">
2279                                <p>
2280                                    Mandarin T3 Production by Chinese and Japanese Native Speakers
2281                                    <br>
2282                                    <span class="w3-text w3-text-theme">
2283                                        Qi Wu
2284                                    </span>
2285                                </p>
2286                            </a>
2287                        </div>
2288                    </div>
2289                    <br>
2290                    <div class="w3-content" style="height:10px"  id="Corpora-based Approaches in Automatic Emotion Recognition"></div>
2291                    <div class="w3-card w3-round w3-white w3-padding">
2292                        <div class="w3-container"  style="margin-top:40px">
2293                            <h4 class="w3-center">Corpora-based Approaches in Automatic Emotion Recognition</h4>
2294                            <hr>
2295                            <a class="w3-text" href="ranjan24_interspeech.html">
2296                                <p>
2297                                    Reinforcement Learning based Data Augmentation for Noise Robust Speech Emotion Recognition
2298                                    <br>
2299                                    <span class="w3-text w3-text-theme">
2300                                        Sumit Ranjan, Rupayan Chakraborty, Sunil Kumar Kopparapu
2301                                    </span>
2302                                </p>
2303                            </a>
2304                            <a class="w3-text" href="mote24_interspeech.html">
2305                                <p>
2306                                    Unsupervised Domain Adaptation for Speech Emotion Recognition using K-Nearest Neighbors Voice Conversion
2307                                    <br>
2308                                    <span class="w3-text w3-text-theme">
2309                                        Pravin Mote, Berrak Sisman, Carlos Busso
2310                                    </span>
2311                                </p>
2312                            </a>
2313                            <a class="w3-text" href="wang24ja_interspeech.html">
2314                                <p>
2315                                    Confidence-aware Hypothesis Transfer Networks for Source-Free Cross-Corpus Speech Emotion Recognition
2316                                    <br>
2317                                    <span class="w3-text w3-text-theme">
2318                                        Jincen Wang, Yan Zhao, Cheng Lu, Hailun Lian, Hongli Chang, Yuan Zong, Wenming Zheng
2319                                    </span>
2320                                </p>
2321                            </a>
2322                            <a class="w3-text" href="xi24_interspeech.html">
2323                                <p>
2324                                    An Effective Local Prototypical Mapping Network for Speech Emotion Recognition
2325                                    <br>
2326                                    <span class="w3-text w3-text-theme">
2327                                        Yuxuan Xi, Yan Song, Lirong Dai, Haoyu Song, Ian McLoughlin
2328                                    </span>
2329                                </p>
2330                            </a>
2331                            <a class="w3-text" href="gao24f_interspeech.html">
2332                                <p>
2333                                    Speech Emotion Recognition with Multi-level Acoustic and 
2333Semantic Information Extraction and Interaction
2334                                    <br>
2335                                    <span class="w3-text w3-text-theme">
2336                                        Yuan Gao, Hao Shi, Chenhui Chu, Tatsuya Kawahara
2337                                    </span>
2338                                </p>
2339                            </a>
2340                        </div>
2341                    </div>
2342                    <br>
2343                    <div class="w3-content" style="height:10px"  id="Analysis of Speakers States and Traits"></div>
2344                    <div class="w3-card w3-round w3-white w3-padding">
2345                        <div class="w3-container"  style="margin-top:40px">
2346                            <h4 class="w3-center">Analysis of Speakers States and Traits</h4>
2347                            <hr>
2348                            <a class="w3-text" href="niebuhr24_interspeech.html">
2349                                <p>
2350                                    How rhythm metrics are linked to produced and perceived speaker charisma
2351                                    <br>
2352                                    <span class="w3-text w3-text-theme">
2353                                        Oliver Niebuhr, Nafiseh Taghva
2354                                    </span>
2355                                </p>
2356                            </a>
2357                            <a class="w3-text" href="li24ma_interspeech.html">
2358                                <p>
2359                                    A Functional Trade-off between Prosodic and Semantic Cues in Conveying Sarcasm
2360                                    <br>
2361                                    <span class="w3-text w3-text-theme">
2362                                        Zhu Li, Xiyuan Gao, Yuqing Zhang, Shekhar Nayak, Matt Coler
2363                                    </span>
2364                                </p>
2365                            </a>
2366                            <a class="w3-text" href="murzaku24_interspeech.html">
2367                                <p>
2368                                    Multimodal Belief Prediction
2369                                    <br>
2370                                    <span class="w3-text w3-text-theme">
2371                                        John Murzaku, Adil Soubki, Owen Rambow
2372                                    </span>
2373                                </p>
2374                            </a>
2375                            <a class="w3-text" href="chen24f_interspeech.html">
2376                                <p>
2377                                    Detecting Empathy in Speech
2378                                    <br>
2379                                    <span class="w3-text w3-text-theme">
2380                                        Run Chen, Haozhe Chen, Anushka Kulkarni, Eleanor Lin, Linda Pang, Divya Tadimeti, Jun Shin, Julia Hirschberg
2381                                    </span>
2382                                </p>
2383                            </a>
2384                            <a class="w3-text" href="tao24b_interspeech.html">
2385                                <p>
2386                                    Learning Representation of Therapist Empathy in Counseling Conversation Using Siamese Hierarchical Attention Network
2387                                    <br>
2388                                    <span class="w3-text w3-text-theme">
2389                                        Dehua Tao, Tan Lee, Harold Chui, Sarah Luk
2390                                    </span>
2391                                </p>
2392                            </a>
2393                            <a class="w3-text" href="kunmei24_interspeech.html">
2394                                <p>
2395                                    Modelling Lexical Characteristics of the Healthy Aging Population: A Corpus-Based Study
2396                                    <br>
2397                                    <span class="w3-text w3-text-theme">
2398                                        Han Kunmei
2399                                    </span>
2400                                </p>
2401                            </a>
2402                            <a class="w3-text" href="gerczuk24_interspeech.html">
2403                                <p>
2404                                    Exploring Gender-Specific Speech Patterns in Automatic Suicide Risk Assessment
2405                                    <br>
2406                                    <span class="w3-text w3-text-theme">
2407                                        Maurice Gerczuk, Shahin Amiriparian, Justina Lutz, Wolfgang Strube, Irina Papazova, Alkomiet Hasan, Björn W. Schuller
2408                                    </span>
2409                                </p>
2410                            </a>
2411                        </div>
2412                    </div>
2413                    <br>
2414                    <div class="w3-content" style="height:10px"  id="Spoofing and Deepfake Detection"></div>
2415                    <div class="w3-card w3-round w3-white w3-padding">
2416                        <div class="w3-container"  style="margin-top:40px">
2417                            <h4 class="w3-center">Spoofing and Deepfake Detection</h4>
2418                            <hr>
2419                            <a class="w3-text" href="klein24_interspeech.html">
2420                                <p>
2421                                    Source Tracing of Audio Deepfake Systems
2422                                    <br>
2423                                    <span class="w3-text w3-text-theme">
2424                                        Nicholas Klein, Tianxiang Chen, Hemlata Tak, Ricardo Casal, Elie Khoury
2425                                    </span>
2426                                </p>
2427                            </a>
2428                            <a class="w3-text" href="liu24m_interspeech.html">
2429                                <p>
2430                                    How Do Neural Spoofing Countermeasures Detect Partially Spoofed Audio?
2431                                    <br>
2432                                    <span class="w3-text w3-text-theme">
2433                                        Tianchi Liu, Lin Zhang, Rohan Kumar Das, Yi Ma, Ruijie Tao, Haizhou Li
2434                                    </span>
2435                                </p>
2436                            </a>
2437                            <a class="w3-text" href="wang24l_interspeech.html">
2438                                <p>
2439                                    Revisiting and Improving Scoring Fusion for Spoofing-aware Speaker Verification Using Compositional Data Analysis
2440                                    <br>
2441                                    <span class="w3-text w3-text-theme">
2442                                        Xin Wang, Tomi Kinnunen, Kong Aik Lee, Paul-Gauthier Noé, Junichi Yamagishi
2443                                    </span>
2444                                </p>
2445                            </a>
2446                            <a class="w3-text" href="baser24_interspeech.html">
2447                                <p>
2448                                    SecureSpectra: Safeguarding Digital Identity from Deep Fake Threats via Intelligent Signatures
2449                                    <br>
2450                                    <span class="w3-text w3-text-theme">
2451                                        Oguzhan Baser, Kaan Kale, Sandeep P. Chinchali
2452                                    </span>
2453                                </p>
2454                            </a>
2455                            <a class="w3-text" href="li24oa_interspeech.html">
2456                                <p>
2457                                    Interpretable Temporal Class Activation Representation for Audio Spoofing  Detection
2458                                    <br>
2459                                    <span class="w3-text w3-text-theme">
2460                                        Menglu Li, Xiao-Ping Zhang
2461                                    </span>
2462                                </p>
2463                            </a>
2464                            <a class="w3-text" href="ge24_interspeech.html">
2465                                <p>
2466                                    DGPN: A Dual Graph Prototypical Network for Few-Shot Speech Spoofing Algorithm Recognition
2467                                    <br>
2468                                    <span class="w3-text w3-text-theme">
2469                                        Zirui Ge, Xinzhou Xu, Haiyan Guo, Tingting Wang, Zhen Yang, Björn W. Schuller
2470                                    </span>
2471                                </p>
2472                            </a>
2473                        </div>
2474                    </div>
2475                    <br>
2476                    <div class="w3-content" style="height:10px"  id="Audio Captioning, Tagging, and Audio-Text Retrieval"></div>
2477                    <div class="w3-card w3-round w3-white w3-padding">
2478                        <div class="w3-container"  style="margin-top:40px">
2479                            <h4 class="w3-center">Audio Captioning, Tagging, and Audio-Text Retrieval</h4>
2480                            <hr>
2481                            <a class="w3-text" href="sun24c_interspeech.html">
2482                                <p>
2483                                    PFCA-Net: Pyramid Feature Fusion and Cross Content Attention Network for Automated Audio Captioning
2484                                    <br>
2485                                    <span class="w3-text w3-text-theme">
2486                                        Jianyuan Sun, Wenwu Wang, Mark D. Plumbley
2487                                    </span>
2488                                </p>
2489                            </a>
2490                            <a class="w3-text" href="liu24_interspeech.html">
2491                                <p>
2492                                    Enhancing Automated Audio Captioning via Large Language Models with Optimized Audio Encoding
2493                                    <br>
2494                                    <span class="w3-text w3-text-theme">
2495                                        Jizhong Liu, Gang Li, Junbo Zhang, Heinrich Dinkel, Yongqing Wang, Zhiyong Yan, Yujun Wang, Bin Wang
2496                                    </span>
2497                                </p>
2498                            </a>
2499                            <a class="w3-text" href="xin24b_interspeech.html">
2500                                <p>
2501                                    Audio-text Retrieval with Transformer-based Hierarchical Alignment and Disentangled Cross-modal Representation
2502                                    <br>
2503                                    <span class="w3-text w3-text-theme">
2504                                        Yifei Xin, Zhihong Zhu, Xuxin Cheng, Xusheng Yang, Yuexian Zou
2505                                    </span>
2506                                </p>
2507                            </a>
2508                            <a class="w3-text" href="dinkel24_interspeech.html">
2509                                <p>
2510                                    Streaming Audio Transformers for Online Audio Tagging
2511                                    <br>
2512                                    <span class="w3-text w3-text-theme">
2513                                        Heinrich Dinkel, Zhiyong Yan, Yongqing Wang, Junbo Zhang, Yujun Wang, Bin Wang
2514                                    </span>
2515                                </p>
2516                            </a>
2517                            <a class="w3-text" href="chaudhary24_interspeech.html">
2518                                <p>
2519                                    Efficient CNNs with Quaternion Transformations and Pruning for Audio Tagging
2520                                    <br>
2521                                    <span class="w3-text w3-text-theme">
2522                                        Aryan Chaudhary, Arshdeep Singh, Vinayak Abrol, Mark D. Plumbley
2523                                    </span>
2524                                </p>
2525                            </a>
2526                            <a class="w3-text" href="jing24b_interspeech.html">
2527                                <p>
2528                                    ParaCLAP – Towards a general language-audio model for computational paralinguistic tasks
2529                                    <br>
2530                                    <span class="w3-text w3-text-theme">
2531                                        Xin Jing, Andreas Triantafyllopoulos, Björn Schuller
2532                                    </span>
2533                                </p>
2534                            </a>
2535                            <a class="w3-text" href="xu24e_interspeech.html">
2536                                <p>
2537                                    Efficient Audio Captioning with Encoder-Level Knowledge Distillation
2538                                    <br>
2539                                    <span class="w3-text w3-text-theme">
2540                                        Xuenan Xu, Haohe Liu, Mengyue Wu, Wenwu Wang, Mark D. Plumbley
2541                                    </span>
2542                                </p>
2543                            </a>
2544                        </div>
2545                    </div>
2546                    <br>
2547                    <div class="w3-content" style="height:10px"  id="Generative Speech Enhancement"></div>
2548                    <div class="w3-card w3-round w3-white w3-padding">
2549                        <div class="w3-container"  style="margin-top:40px">
2550                            <h4 class="w3-center">Generative Speech Enhancement</h4>
2551                            <hr>
2552                            <a class="w3-text" href="scheibler24_interspeech.html">
2553                                <p>
2554                                    Universal Score-based Speech Enhancement with High Content Preservation
2555                                    <br>
2556                                    <span class="w3-text w3-text-theme">
2557                                        Robin Scheibler, Yusuke Fujita, Yuma Shirahata, Tatsuya Komatsu
2558                                    </span>
2559                                </p>
2560                            </a>
2561                            <a class="w3-text" href="yang24h_interspeech.html">
2562                                <p>
2563                                    Genhancer: High-Fidelity Speech Enhancement via Generative Modeling on Discrete Codec Tokens
2564                                    <br>
2565                                    <span class="w3-text w3-text-theme">
2566                                        Haici Yang, Jiaqi Su, Minje Kim, Zeyu Jin
2567                                    </span>
2568                                </p>
2569                            </a>
2570                            <a class="w3-text" href="jukic24_interspeech.html">
2571                                <p>
2572                                    Schrödinger Bridge for Generative Speech Enhancement
2573                                    <br>
2574                                    <span class="w3-text w3-text-theme">
2575                                        Ante Jukić, Roman Korostik, Jagadeesh Balam, Boris Ginsburg
2576                                    </span>
2577                                </p>
2578                            </a>
2579                            <a class="w3-text" href="trachu24_interspeech.html">
2580                                <p>
2581                                    Thunder : Unified Regression-Diffusion Speech Enhancement with a Single Reverse Step using Brownian Bridge
2582                                    <br>
2583                                    <span class="w3-text w3-text-theme">
2584                                        Thanapat Trachu, Chawan Piansaddhayanon, Ekapol Chuangsuwanich
2585                                    </span>
2586                                </p>
2587                            </a>
2588                            <a class="w3-text" href="yang24k_interspeech.html">
2589                                <p>
2590                                    Pre-training Feature Guided Diffusion Model for Speech Enhancement
2591                                    <br>
2592                                    <span class="w3-text w3-text-theme">
2593                                        Yiyuan Yang, Niki Trigoni, Andrew Markham
2594                                    </span>
2595                                </p>
2596                            </a>
2597                            <a class="w3-text" href="kim24o_interspeech.html">
2598                                <p>
2599                                    Guided conditioning with predictive network on score-based diffusion model for speech enhancement
2600                                    <br>
2601                                    <span class="w3-text w3-text-theme">
2602                                        Dail Kim, Da-Hee Yang, Donghyun Kim, Joon-Hyuk Chang, Jeonghwan Choi, Moa Lee, Jaemo Yang, Han-gil Moon
2603                                    </span>
2604                                </p>
2605                            </a>
2606                        </div>
2607                    </div>
2608                    <br>
2609                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Evaluation"></div>
2610                    <div class="w3-card w3-round w3-white w3-padding">
2611                        <div class="w3-container"  style="margin-top:40px">
2612                            <h4 class="w3-center">Speech Synthesis: Evaluation</h4>
2613                            <hr>
2614                            <a class="w3-text" href="yin24b_interspeech.html">
2615                                <p>
2616                                    SVSNet+: Enhancing Speaker Voice Similarity Assessment Models with Representations from Speech Foundation Models
2617                                    <br>
2618                                    <span class="w3-text w3-text-theme">
2619                                        Chun Yin, Tai-Shih Chi, Yu Tsao, Hsin-Min Wang
2620                                    </span>
2621                                </p>
2622                            </a>
2623                            <a class="w3-text" href="anand24_interspeech.html">
2624                                <p>
2625                                    Enhancing Out-of-Vocabulary Performance of Indian TTS Systems for Practical Applications through Low-Effort Data Strategies
2626                                    <br>
2627                                    <span class="w3-text w3-text-theme">
2628                                        Srija Anand, Praveen Srinivasa Varadhan, Ashwin Sankar, Giri Raju, Mitesh M. Khapra
2629                                    </span>
2630                                </p>
2631                            </a>
2632                            <a class="w3-text" href="edlund24_interspeech.html">
2633                                <p>
2634                                    Assessing the impact of contextual framing on subjective TTS quality
2635                                    <br>
2636                                    <span class="w3-text w3-text-theme">
2637                                        Jens Edlund, Christina Tånnander, Sébastien Le Maguer, Petra Wagner
2638                                    </span>
2639                                </p>
2640                            </a>
2641                            <a class="w3-text" href="adigwe24_interspeech.html">
2642                                <p>
2643                                    What do people hear? Listeners’ Perception of Conversational Speech
2644                                    <br>
2645                                    <span class="w3-text w3-text-theme">
2646                                        Adaeze Adigwe, Sarenne Wallbridge, Simon King
2647                                    </span>
2648                                </p>
2649                            </a>
2650                            <a class="w3-text" href="wang24s_interspeech.html">
2651                                <p>
2652                                    Uncertainty-Aware Mean Opinion Score Prediction
2653                                    <br>
2654                                    <span class="w3-text w3-text-theme">
2655                                        Hui Wang, Shiwan Zhao, Jiaming Zhou, Xiguang Zheng, Haoqin Sun, Xuechen Wang, Yong Qin
2656                                    </span>
2657                                </p>
2658                            </a>
2659                            <a class="w3-text" href="saget24_interspeech.html">
2660                                <p>
2661                                    Lifelong Learning MOS Prediction for Synthetic Speech Quality Evaluation
2662                                    <br>
2663                                    <span class="w3-text w3-text-theme">
2664                                        Félix Saget, Meysam Shamsi, Marie Tahon
2665                                    </span>
2666                                </p>
2667                            </a>
2668                        </div>
2669                    </div>
2670                    <br>
2671                    <div class="w3-content" style="height:10px"  id="Multilingual ASR"></div>
2672                    <div class="w3-card w3-round w3-white w3-padding">
2673                        <div class="w3-container"  style="margin-top:40px">
2674                            <h4 class="w3-center">Multilingual ASR</h4>
2675                            <hr>
2676                            <a class="w3-text" href="kwok24_interspeech.html">
2677                                <p>
2678                                    Continual Learning Optimizations for Auto-regressive Decoder of Multilingual ASR systems
2679                                    <br>
2680                                    <span class="w3-text w3-text-theme">
2681                                        Chin Yuen Kwok, Jia Qi Yip, Eng Siong Chng
2682                                    </span>
2683                                </p>
2684                            </a>
2685                            <a class="w3-text" href="shi24g_interspeech.html">
2686                                <p>
2687                                    ML-SUPERB 2.0: Benchmarking Multilingual Speech Models Across Modeling Constraints, Languages, and Datasets
2688                                    <br>
2689                                    <span class="w3-text w3-text-theme">
2690                                        Jiatong Shi, Shih-Heng Wang, William Chen, Martijn Bartelds, Vanya Bannihatti Kumar, Jinchuan Tian, Xuankai Chang, Dan Jurafsky, Karen Livescu, Hung-yi Lee, Shinji Watanabe
2691                                    </span>
2692                                </p>
2693                            </a>
2694                            <a class="w3-text" href="pineiromartin24_interspeech.html">
2695                                <p>
2696                                    Weighted Cross-entropy for Low-Resource Languages in Multilingual Speech Recognition
2697                                    <br>
2698                                    <span class="w3-text w3-text-theme">
2699                                        Andrés Piñeiro-Martín, Carmen García-Mateo, Laura Docio-Fernandez, María del Carmen López-Pérez, Georg Rehm
2700                                    </span>
2701                                </p>
2702                            </a>
2703                            <a class="w3-text" href="saif24_interspeech.html">
2704                                <p>
2705                                    M2ASR: Multilingual Multi-task Automatic Speech Recognition via Multi-objective Optimization
2706                                    <br>
2707                                    <span class="w3-text w3-text-theme">
2708                                        A F M Saif, Lisha Chen, Xiaodong Cui, Songtao Lu, Brian Kingsbury, Tianyi Chen
2709                                    </span>
2710                                </p>
2711                            </a>
2712                            <a class="w3-text" href="li24s_interspeech.html">
2713                                <p>
2714                                    MSR-86K: An Evolving, Multilingual Corpus with 86,300 Hours of Transcribed Audio for Speech Recognition Research
2715                                    <br>
2716                                    <span class="w3-text w3-text-theme">
2717                                        Song Li, Yongbin You, Xuezhi Wang, Zhengkun Tian, Ke Ding, Guanglu Wan
2718                                    </span>
2719                                </p>
2720                            </a>
2721                            <a class="w3-text" href="houston24_interspeech.html">
2722                                <p>
2723                                    Improving Multilingual ASR Robustness to Errors in Language Input
2724                                    <br>
2725                                    <span class="w3-text w3-text-theme">
2726                                        Brady Houston, Omid Sadjadi, Zejiang Hou, Srikanth Vishnubhotla, Kyu J. Han
2727                                    </span>
2728                                </p>
2729                            </a>
2730                        </div>
2731                    </div>
2732                    <br>
2733                    <div class="w3-content" style="height:10px"  id="General Topics in ASR"></div>
2734                    <div class="w3-card w3-round w3-white w3-padding">
2735                        <div class="w3-container"  style="margin-top:40px">
2736                            <h4 class="w3-center">General Topics in ASR</h4>
2737                            <hr>
2738                            <a class="w3-text" href="suh24_interspeech.html">
2739                                <p>
2740                                    Improving Domain-Specific ASR with LLM-Generated Contextual Descriptions
2741                                    <br>
2742                                    <span class="w3-text w3-text-theme">
2743                                        Jiwon Suh, Injae Na, Woohwan Jung
2744                                    </span>
2745                                </p>
2746                            </a>
2747                            <a class="w3-text" href="li24c_interspeech.html">
2748                                <p>
2749                                    A Multitask Training Approach to Enhance Whisper with Open-Vocabulary Keyword Spotting
2750                                    <br>
2751                                    <span class="w3-text w3-text-theme">
2752                                        Yuang Li, Min Zhang, Chang Su, Yinglu Li, Xiaosong Qiao, Mengxin Ren, Miaomiao Ma, Daimeng Wei, Shimin Tao, Hao Yang
2753                                    </span>
2754                                </p>
2755                            </a>
2756                            <a class="w3-text" href="zusag24_interspeech.html">
2757                                <p>
2758                                    CrisperWhisper: Accurate Timestamps on Verbatim Speech Transcriptions
2759                                    <br>
2760                                    <span class="w3-text w3-text-theme">
2761                                        Mario Zusag, Laurin Wagner, Bernhad Thallinger
2762                                    </span>
2763                                </p>
2764                            </a>
2765                            <a class="w3-text" href="mihajlik24_interspeech.html">
2766                                <p>
2767                                    On Disfluency and Non-lexical Sound Labeling for End-to-end Automatic Speech Recognition
2768                                    <br>
2769                                    <span class="w3-text w3-text-theme">
2770                                        Peter Mihajlik, Yan Meng, Mate S Kadar, Julian Linke, Barbara Schuppler, Katalin Mády
2771                                    </span>
2772                                </p>
2773                            </a>
2774                            <a class="w3-text" href="mujtaba24_interspeech.html">
2775                                <p>
2776                                    Inclusive ASR for Disfluent Speech: Cascaded Large-Scale Self-Supervised Learning with Targeted Fine-Tuning and Data Augmentation
2777                                    <br>
2778                                    <span class="w3-text w3-text-theme">
2779                                        Dena Mujtaba, Nihar R. Mahapatra, Megan Arney, J. Scott Yaruss, Caryn Herring, Jia Bin
2780                                    </span>
2781                                </p>
2782                            </a>
2783                            <a class="w3-text" href="tan24_interspeech.html">
2784                                <p>
2785                                    DualPure: An Efficient Adversarial Purification Method for Speech Command Recognition
2786                                    <br>
2787                                    <span class="w3-text w3-text-theme">
2788                                        Hao Tan, Xiaochen Liu, Huan Zhang, Junjian Zhang, Yaguan Qian, Zhaoquan Gu
2789                                    </span>
2790                                </p>
2791                            </a>
2792                            <a class="w3-text" href="lehecka24_interspeech.html">
2793                                <p>
2794                                    A Comparative Analysis of Bilingual and Trilingual Wav2Vec Models for Automatic Speech Recognition in Multilingual Oral History Archives
2795                                    <br>
2796                                    <span class="w3-text w3-text-theme">
2797                                        Jan Lehečka, Josef V. Psutka, Lubos Smidl, Pavel Ircing, Josef Psutka
2798                                    </span>
2799                                </p>
2800                            </a>
2801                            <a class="w3-text" href="delafuente24_interspeech.html">
2802                                <p>
2803                                    A layer-wise analysis of Mandarin and English suprasegmentals in SSL speech models
2804                                    <br>
2805                                    <span class="w3-text w3-text-theme">
2806                                        Anton de la Fuente, Dan Jurafsky
2807                                    </span>
2808                                </p>
2809                            </a>
2810                            <a class="w3-text" href="leivaditi24_interspeech.html">
2811                                <p>
2812                                    Fine-Tuning Strategies for Dutch Dysarthric Speech Recognition: Evaluating the Impact of Healthy, Disease-Specific, and Speaker-Specific Data
2813                                    <br>
2814                                    <span class="w3-text w3-text-theme">
2815                                        Spyretta Leivaditi, Tatsunari Matsushima, Matt Coler, Shekhar Nayak, Vass Verkhodanova
2816                                    </span>
2817                                </p>
2818                            </a>
2819                            <a class="w3-text" href="hsieh24_interspeech.html">
2820                                <p>
2821                                    Dysarthric Speech Recognition Using Curriculum Learning and Articulatory Feature Embedding
2822                                    <br>
2823                                    <span class="w3-text w3-text-theme">
2824                                        I-Ting Hsieh, Chung-Hsien Wu
2825                                    </span>
2826                                </p>
2827                            </a>
2828                            <a class="w3-text" href="wang24x_interspeech.html">
2829                                <p>
2830                                    Enhancing Dysarthric Speech Recognition for Unseen Speakers via Prototype-Based Adaptation
2831                                    <br>
2832                                    <span class="w3-text w3-text-theme">
2833                                        Shiyao Wang, Shiwan Zhao, Jiaming Zhou, Aobo Kong, Yong Qin
2834                                    </span>
2835                                </p>
2836                            </a>
2837                            <a class="w3-text" href="zheng24_interspeech.html">
2838                                <p>
2839                                    An efficient text augmentation approach for contextualized Mandarin speech recognition
2840                                    <br>
2841                                    <span class="w3-text w3-text-theme">
2842                                        Naijun Zheng, Xucheng Wan, Kai Liu, Ziqing Du, Zhou Huan
2843                                    </span>
2844                                </p>
2845                            </a>
2846                            <a class="w3-text" href="li24h_interspeech.html">
2847                                <p>
2848                                    Investigating ASR Error Correction with Large Language Model and Multilingual 1-best Hypotheses
2849                                    <br>
2850                                    <span class="w3-text w3-text-theme">
2851                                        Sheng Li, Chen Chen, Chin Yuen Kwok, Chenhui Chu, Eng Siong Chng, Hisashi Kawai
2852                                    </span>
2853                                </p>
2854                            </a>
2855                            <a class="w3-text" href="wang24n_interspeech.html">
2856                                <p>
2857                                    Efficiently Train ASR Models that Memorize Less and Perform Better with Per-core Clipping
2858                                    <br>
2859                                    <span class="w3-text w3-text-theme">
2860                                        Lun Wang, Om Thakkar, Zhong Meng, Nicole Rafidi, Rohit Prabhavalkar, Arun Narayanan
2861                                    </span>
2862                                </p>
2863                            </a>
2864                        </div>
2865                    </div>
2866                    <br>
2867                    <div class="w3-content" style="height:10px"  id="Spoken Language Understanding"></div>
2868                    <div class="w3-card w3-round w3-white w3-padding">
2869                        <div class="w3-container"  style="margin-top:40px">
2870                            <h4 class="w3-center">Spoken Language Understanding</h4>
2871                            <hr>
2872                            <a class="w3-text" href="phung24_interspeech.html">
2873                                <p>
2874                                    AR-NLU: A Framework for Enhancing Natural Language Understanding Model Robustness against ASR Errors
2875                                    <br>
2876                                    <span class="w3-text w3-text-theme">
2877                                        Emmy Phung, Harsh Deshpande, Ahmad Emami, Kanishk Singh
2878                                    </span>
2879                                </p>
2880                            </a>
2881                            <a class="w3-text" href="li24_interspeech.html">
2882                                <p>
2883                                    Prompting Whisper for QA-driven Zero-shot End-to-end Spoken Language Understanding
2884                                    <br>
2885                                    <span class="w3-text w3-text-theme">
2886                                        Mohan Li, Simon Keizer, Rama Doddipatla
2887                                    </span>
2888                                </p>
2889                            </a>
2890                            <a class="w3-text" href="tran24b_interspeech.html">
2891                                <p>
2892                                    VN-SLU: A Vietnamese Spoken Language Understanding Dataset
2893                                    <br>
2894                                    <span class="w3-text w3-text-theme">
2895                                        Tuyen Tran, Khanh Le, Ngoc Dang Nguyen, Minh Vu, Huyen Ngo, Woomyoung Park, Thi Thu Trang Nguyen
2896                                    </span>
2897                                </p>
2898                            </a>
2899                            <a class="w3-text" href="kando24_interspeech.html">
2900                                <p>
2901                                    Textless Dependency Parsing by Labeled Sequence Prediction
2902                                    <br>
2903                                    <span class="w3-text w3-text-theme">
2904                                        Shunsuke Kando, Yusuke Miyao, Jason Naradowsky, Shinnosuke Takamichi
2905                                    </span>
2906                                </p>
2907                            </a>
2908                            <a class="w3-text" href="yue24_interspeech.html">
2909                                <p>
2910                                    Towards Speech Classification from Acoustic and Vocal Tract data in Real-time MRI
2911                                    <br>
2912                                    <span class="w3-text w3-text-theme">
2913                                        Yaoyao Yue, Michael Proctor, Luping Zhou, Rijul Gupta, Tharinda Piyadasa, Amelia Gully, Kirrie Ballard, Craig Jin
2914                                    </span>
2915                                </p>
2916                            </a>
2917                            <a class="w3-text" href="johnson24_interspeech.html">
2918                                <p>
2919                                    Efficient SQA from Long Audio Contexts: A Policy-driven Approach
2920                                    <br>
2921                                    <span class="w3-text w3-text-theme">
2922                                        Alexander Johnson, Peter Plantinga, Pheobe Sun, Swaroop Gadiyaram, Abenezer Girma, Ahmad Emami
2923                                    </span>
2924                                </p>
2925                            </a>
2926                        </div>
2927                    </div>
2928                    <br>
2929                    <div class="w3-content" style="height:10px"  id="Speech and Multimodal Resources"></div>
2930                    <div class="w3-card w3-round w3-white w3-padding">
2931                        <div class="w3-container"  style="margin-top:40px">
2932                            <h4 class="w3-center">Speech and Multimodal Resources</h4>
2933                            <hr>
2934                            <a class="w3-text" href="pesan24_interspeech.html">
2935                                <p>
2936                                    BESST Dataset: A Multimodal Resource for Speech-based Stress Detection and Analysis
2937                                    <br>
2938                                    <span class="w3-text w3-text-theme">
2939                                        Jan Pešán, Vojtěch Juřík, Martin Karafiát, Jan Černocký
2940                                    </span>
2941                                </p>
2942                            </a>
2943                            <a class="w3-text" href="turetzky24_interspeech.html">
2944                                <p>
2945                                    HebDB: a Weakly Supervised Dataset for Hebrew Speech Processing
2946                                    <br>
2947                                    <span class="w3-text w3-text-theme">
2948                                        Arnon Turetzky, Or Tal, Yael Segal, Yehoshua Dissen, Ella Zeldes, Amit Roth, Eyal Cohen, Yosi Shrem, Bronya R. Chernyak, Olga Seleznova, Joseph Keshet, Yossi Adi
2949                                    </span>
2950                                </p>
2951                            </a>
2952                            <a class="w3-text" href="wang24b_interspeech.html">
2953                                <p>
2954                                    GLOBE: A High-quality English Corpus with Global Accents for Zero-shot Speaker Adaptive Text-to-Speech
2955                                    <br>
2956                                    <span class="w3-text w3-text-theme">
2957                                        Wenbin Wang, Yang Song, Sanjay Jha
2958                                    </span>
2959                                </p>
2960                            </a>
2961                            <a class="w3-text" href="kong24_interspeech.html">
2962                                <p>
2963                                    STraDa: A Singer Traits Dataset
2964                                    <br>
2965                                    <span class="w3-text w3-text-theme">
2966                                        Yuexuan Kong, Viet-Anh Tran, Romain Hennequin
2967                                    </span>
2968                                </p>
2969                            </a>
2970                            <a class="w3-text" href="anderer24_interspeech.html">
2971                                <p>
2972                                    MaViLS, a Benchmark Dataset for Video-to-Slide Alignment, Assessing Baseline Accuracy with a Multimodal Alignment Algorithm Leveraging Speech, OCR, and Visual Features
2973                                    <br>
2974                                    <span class="w3-text w3-text-theme">
2975                                        Katharina Anderer, Andreas Reich, Matthias Wölfel
2976                                    </span>
2977                                </p>
2978                            </a>
2979                            <a class="w3-text" href="sungbin24_interspeech.html">
2980                                <p>
2981                                    MultiTalk: Enhancing 3D Talking Head Generation Across Languages with Multilingual Video Dataset
2982                                    <br>
2983                                    <span class="w3-text w3-text-theme">
2984                                        Kim Sung-Bin, Lee Chae-Yeon, Gihun Son, Oh Hyun-Bin, Janghoon Ju, Suekyeong Nam, Tae-Hyun Oh
2985                                    </span>
2986                                </p>
2987                            </a>
2988                            <a class="w3-text" href="veliche24_interspeech.html">
2989                                <p>
2990                                    Towards measuring fairness in speech recognition: Fair-Speech dataset
2991                                    <br>
2992                                    <span class="w3-text w3-text-theme">
2993                                        Irina-Elena Veliche, Zhuangqun Huang, Vineeth Ayyat Kochaniyan, Fuchun Peng, Ozlem Kalinli, Michael L. Seltzer
2994                                    </span>
2995                                </p>
2996                            </a>
2997                            <a class="w3-text" href="lu24f_interspeech.html">
2998                                <p>
2999                                    Codecfake: An Initial Dataset for Detecting LLM-based Deepfake Audio
3000                                    <br>
3001                                    <span class="w3-text w3-text-theme">
3002                                        Yi Lu, Yuankun Xie, Ruibo Fu, Zhengqi Wen, Jianhua Tao, Zhiyong Wang, Xin Qi, Xuefei Liu, Yongwei Li, Yukun Liu, Xiaopeng Wang, Shuchen Shi
3003                                    </span>
3004                                </p>
3005                            </a>
3006                            <a class="w3-text" href="osman24_interspeech.html">
3007                                <p>
3008                                    SER Evals: In-domain and Out-of-domain benchmarking for speech emotion recognition
3009                                    <br>
3010                                    <span class="w3-text w3-text-theme">
3011                                        Mohamed Osman, Daniel Z. Kaplan, Tamer Nadeem
3012                                    </span>
3013                                </p>
3014                            </a>
3015                        </div>
3016                    </div>
3017                    <br>
3018                    <div class="w3-content" style="height:10px"  id="Pathological Speech Analysis 1"></div>
3019                    <div class="w3-card w3-round w3-white w3-padding">
3020                        <div class="w3-container"  style="margin-top:40px">
3021                            <h4 class="w3-center">Pathological Speech Analysis 1</h4>
3022                            <hr>
3023                            <a class="w3-text" href="gudmundsson24_interspeech.html">
3024                                <p>
3025                                    The MARRYS helmet: A new device for researching and training “jaw dancing”
3026                                    <br>
3027                                    <span class="w3-text w3-text-theme">
3028                                        Vidar Freyr Gudmundsson, Keve Márton Gönczi, Malin Svensson Lundmark, Donna Erickson, Oliver Niebuhr
3029                                    </span>
3030                                </p>
3031                            </a>
3032                            <a class="w3-text" href="laquatra24_interspeech.html">
3033                                <p>
3034                                    Exploiting Foundation Models and Speech Enhancement for Parkinson's Disease Detection from Speech in Real-World Operative Conditions
3035                                    <br>
3036                                    <span class="w3-text w3-text-theme">
3037                                        Moreno La Quatra, Maria Francesca Turco, Torbjørn Svendsen, Giampiero Salvi, Juan Rafael Orozco-Arroyave, Sabato Marco Siniscalchi
3038                                    </span>
3039                                </p>
3040                            </a>
3041                            <a class="w3-text" href="triantafyllopoulos24_interspeech.html">
3042                                <p>
3043                                    Sustained Vowels for Pre- vs Post-Treatment COPD Classification
3044                                    <br>
3045                                    <span class="w3-text w3-text-theme">
3046                                        Andreas Triantafyllopoulos, Anton Batliner, Wolfgang Mayr, Markus Fendler, Florian Pokorny, Maurice Gerczuk, Shahin Amiriparian, Thomas Berghaus, Björn Schuller
3047                                    </span>
3048                                </p>
3049                            </a>
3050                            <a class="w3-text" href="amiri24_interspeech.html">
3051                                <p>
3052                                    Adversarial Robustness Analysis in Automatic Pathological Speech Detection Approaches
3053                                    <br>
3054                                    <span class="w3-text w3-text-theme">
3055                                        Mahdi Amiri, Ina Kodrasi
3056                                    </span>
3057                                </p>
3058                            </a>
3059                            <a class="w3-text" href="kim24q_interspeech.html">
3060                                <p>
3061                                    Automatic Children Speech Sound Disorder Detection with Age and Speaker Bias Mitigation
3062                                    <br>
3063                                    <span class="w3-text w3-text-theme">
3064                                        Gahye Kim, Yunjung Eom, Selina S. Sung, Seunghee Ha, Tae-Jin Yoon, Jungmin So
3065                                    </span>
3066                                </p>
3067                            </a>
3068                        </div>
3069                    </div>
3070                    <br>
3071                    <div class="w3-content" style="height:10px"  id="Speech and Language in Health: from Remote Monitoring to Medical Conversations - 1 (Special Session)"></div>
3072                    <div class="w3-card w3-round w3-white w3-padding">
3073                        <div class="w3-container"  style="margin-top:40px">
3074                            <h4 class="w3-center">Speech and Language in Health: from Remote Monitoring to Medical Conversations - 1 (Special Session)</h4>
3075                            <hr>
3076                            <a class="w3-text" href="kadkhodaieelyaderani24_interspeech.html">
3077                                <p>
3078                                    Reference-Free Estimation of the Quality of Clinical Notes Generated from Doctor-Patient Conversations
3079                                    <br>
3080                                    <span class="w3-text w3-text-theme">
3081                                        Mojtaba Kadkhodaie Elyaderani, John Glover, Thomas Schaaf
3082                                    </span>
3083                                </p>
3084                            </a>
3085                            <a class="w3-text" href="mun24_interspeech.html">
3086                                <p>
3087                                    Developing an End-to-End Framework for Predicting the Social Communication Severity Scores of Children with Autism Spectrum Disorder
3088                                    <br>
3089                                    <span class="w3-text w3-text-theme">
3090                                        Jihyun Mun, Sunhee Kim, Minhwa Chung
3091                                    </span>
3092                                </p>
3093                            </a>
3094                            <a class="w3-text" href="despotovic24_interspeech.html">
3095                                <p>
3096                                    Multimodal Fusion for Vocal Biomarkers Using Vector Cross-Attention
3097                                    <br>
3098                                    <span class="w3-text w3-text-theme">
3099                                        Vladimir Despotovic, Abir Elbéji, Petr V. Nazarov, Guy Fagherazzi
3100                                    </span>
3101                                </p>
3102                            </a>
3103                            <a class="w3-text" href="goria24_interspeech.html">
3104                                <p>
3105                                    Revealing Confounding Biases: A Novel Benchmarking Approach for Aggregate-Level Performance Metrics in Health Assessments
3106                                    <br>
3107                                    <span class="w3-text w3-text-theme">
3108                                        Stefano Goria, Roseline Polle, Salvatore Fara, Nicholas Cummins
3109                                    </span>
3110                                </p>
3111                            </a>
3112                            <a class="w3-text" href="rameau24_interspeech.html">
3113                                <p>
3114                                    Developing Multi-Disorder Voice Protocols: A team science approach involving clinical expertise, bioethics, standards, and DEI.
3115                                    <br>
3116                                    <span class="w3-text w3-text-theme">
3117                                        Anaïs Rameau, Satrajit Ghosh, Alexandros Sigaras, Olivier Elemento, Jean-Christophe Belisle-Pipon, Vardit Ravitsky, Maria Powell, Alistair Johnson, David Dorr, Philip Payne, Micah Boyer, Stephanie Watts, Ruth Bahr, Frank Rudzicz, Jordan Lerner-Ellis, Shaheen Awan, Don Bolser, Yael Bensoussan
3118                                    </span>
3119                                </p>
3120                            </a>
3121                            <a class="w3-text" href="dumpala24b_interspeech.html">
3122                                <p>
3123                                    Self-Supervised Embeddings for Detecting Individual Symptoms of Depression
3124                                    <br>
3125                                    <span class="w3-text w3-text-theme">
3126                                        Sri Harsha Dumpala, Katerina Dikaios, Abraham Nunes, Frank Rudzicz, Rudolf Uher, Sageev Oore
3127                                    </span>
3128                                </p>
3129                            </a>
3130                            <a class="w3-text" href="mehta24_interspeech.html">
3131                                <p>
3132                                    Comparing ambulatory voice measures during daily life with brief laboratory assessments in speakers with and without vocal hyperfunction
3133                                    <br>
3134                                    <span class="w3-text w3-text-theme">
3135                                        Daryush D. Mehta, Jarrad H. Van Stan, Hamzeh Ghasemzadeh, Robert E. Hillman
3136                                    </span>
3137                                </p>
3138                            </a>
3139                            <a class="w3-text" href="williams24_interspeech.html">
3140                                <p>
3141                                    Predicting Acute Pain Levels Implicitly from Vocal Features
3142                                    <br>
3143                                    <span class="w3-text w3-text-theme">
3144                                        Jennifer Williams, Eike Schneiders, Henry Card, Tina Seabrooke, Beatrice Pakenham-Walsh, Tayyaba Azim, Lucy Valls-Reed, Ganesh Vigneswaran, John Robert Bautista, Rohan Chandra, Arya Farahi
3145                                    </span>
3146                                </p>
3147                            </a>
3148                            <a class="w3-text" href="demir24_interspeech.html">
3149                                <p>
3150                                    Towards Intelligent Speech Assistants in Operating Rooms: A Multimodal Model for Surgical Workflow Analysis
3151                                    <br>
3152                                    <span class="w3-text w3-text-theme">
3153                                        Kubilay Can Demir, Belén Lojo Rodríguez, Tobias Weise, Andreas Maier, Seung Hee Yang
3154                                    </span>
3155                                </p>
3156                            </a>
3157                            <a class="w3-text" href="premananth24_interspeech.html">
3158                                <p>
3159                                    A Multimodal Framework for the Assessment of the Schizophrenia Spectrum
3160                                    <br>
3161                                    <span class="w3-text w3-text-theme">
3162                                        Gowtham Premananth, Yashish M. Siriwardena, Philip Resnik, Sonia Bansal, Deanna L.Kelly, Carol Espy-Wilson
3163                                    </span>
3164                                </p>
3165                            </a>
3166                        </div>
3167                    </div>
3168                    <br>
3169                    <div class="w3-content" style="height:10px"  id="Speech and Brain"></div>
3170                    <div class="w3-card w3-round w3-white w3-padding">
3171                        <div class="w3-container"  style="margin-top:40px">
3172                            <h4 class="w3-center">Speech and Brain</h4>
3173                            <hr>
3174                            <a class="w3-text" href="wang24ka_interspeech.html">
3175                                <p>
3176                                    Exploring the Complementary Nature of Speech and Eye Movements for Profiling Neurological Disorders
3177                                    <br>
3178                                    <span class="w3-text w3-text-theme">
3179                                        Yuzhe Wang, Anna Favaro, Thomas Thebaud, Jesus Villalba, Najim Dehak, Laureano Moro-Velazquez
3180                                    </span>
3181                                </p>
3182                            </a>
3183                            <a class="w3-text" href="li24l_interspeech.html">
3184                                <p>
3185                                    Refining Self-supervised Learnt Speech Representation using Brain Activations
3186                                    <br>
3187                                    <span class="w3-text w3-text-theme">
3188                                        HengYu Li, Kangdi Mei, Zhaoci Liu, Yang Ai, Liping Chen, Jie Zhang, Zhenhua Ling
3189                                    </span>
3190                                </p>
3191                            </a>
3192                            <a class="w3-text" href="wang24ba_interspeech.html">
3193                                <p>
3194                                    Large Language Model-based FMRI Encoding of Language Functions for Subjects with Neurocognitive Disorder
3195                                    <br>
3196                                    <span class="w3-text w3-text-theme">
3197                                        Yuejiao Wang, Xianmin Gong, Lingwei Meng, Xixin Wu, Helen Meng
3198                                    </span>
3199                                </p>
3200                            </a>
3201                            <a class="w3-text" href="neelabh24_interspeech.html">
3202                                <p>
3203                                    From Sound to Meaning in the Auditory Cortex: A Neuronal Representation and Classification Analysis
3204                                    <br>
3205                                    <span class="w3-text w3-text-theme">
3206                                        Kumar Neelabh, Vishnu Sreekumar
3207                                    </span>
3208                                </p>
3209                            </a>
3210                            <a class="w3-text" href="feng24_interspeech.html">
3211                                <p>
3212                                    Towards an End-to-End Framework for Invasive Brain Signal Decoding with Large Language Models
3213                                    <br>
3214                                    <span class="w3-text w3-text-theme">
3215                                        Sheng Feng, Heyang Liu, Yu Wang, Yanfeng Wang
3216                                    </span>
3217                                </p>
3218                            </a>
3219                            <a class="w3-text" href="lee24c_interspeech.html">
3220                                <p>
3221                                    Toward Fully-End-to-End Listened Speech Decoding from EEG Signals
3222                                    <br>
3223                                    <span class="w3-text w3-text-theme">
3224                                        Jihwan Lee, Aditya Kommineni, Tiantian Feng, Kleanthis Avramidis, Xuan Shi, Sudarsana Reddy Kadiri, Shrikanth Narayanan
3225                                    </span>
3226                                </p>
3227                            </a>
3228                        </div>
3229                    </div>
3230                    <br>
3231                    <div class="w3-content" style="height:10px"  id="Innovative Methods in Phonetics and Phonology"></div>
3232                    <div class="w3-card w3-round w3-white w3-padding">
3233                        <div class="w3-container"  style="margin-top:40px">
3234                            <h4 class="w3-center">Innovative Methods in Phonetics and Phonology</h4>
3235                            <hr>
3236                            <a class="w3-text" href="ahn24d_interspeech.html">
3237                                <p>
3238                                    The Use of Phone Categories and Cross-Language Modeling for Phone Alignment of Panãra
3239                                    <br>
3240                                    <span class="w3-text w3-text-theme">
3241                                        Emily P. Ahn, Eleanor Chodroff, Myriam Lapierre, Gina-Anne Levow
3242                                    </span>
3243                                </p>
3244                            </a>
3245                            <a class="w3-text" href="raybarman24_interspeech.html">
3246                                <p>
3247                                    Deciphering Assamese Vowel Harmony with Featural InfoWaveGAN
3248                                    <br>
3249                                    <span class="w3-text w3-text-theme">
3250                                        Sneha Ray Barman, Shakuntala Mahanta, Neeraj Kumar Sharma
3251                                    </span>
3252                                </p>
3253                            </a>
3254                            <a class="w3-text" href="tadavarthy24_interspeech.html">
3255                                <p>
3256                                    Phonological Feature Detection for US English using the Phonet Library
3257                                    <br>
3258                                    <span class="w3-text w3-text-theme">
3259                                        Harsha Veena Tadavarthy, Austin Jones, Margaret E. L. Renwick
3260                                    </span>
3261                                </p>
3262                            </a>
3263                            <a class="w3-text" href="kaland24_interspeech.html">
3264                                <p>
3265                                    K-means and hierarchical clustering of f0 contours
3266                                    <br>
3267                                    <span class="w3-text w3-text-theme">
3268                                        Constantijn Kaland, Jeremy Steffman, Jennifer Cole
3269                                    </span>
3270                                </p>
3271                            </a>
3272                            <a class="w3-text" href="rousso24_interspeech.html">
3273                                <p>
3274                                    Tradition or Innovation: A Comparison of Modern ASR Methods for Forced Alignment
3275                                    <br>
3276                                    <span class="w3-text w3-text-theme">
3277                                        Rotem Rousso, Eyal Cohen, Joseph Keshet, Eleanor Chodroff
3278                                    </span>
3279                                </p>
3280                            </a>
3281                            <a class="w3-text" href="kim24l_interspeech.html">
3282                                <p>
3283                                    Using wav2vec 2.0 for phonetic classification tasks: methodological aspects
3284                                    <br>
3285                                    <span class="w3-text w3-text-theme">
3286                                        Lila Kim, Cédric Gendrot
3287                                    </span>
3288                                </p>
3289                            </a>
3290                            <a class="w3-text" href="lambropoulos24_interspeech.html">
3291                                <p>
3292                                    The sub-band cepstrum as a tool for locating local spectral regions of phonetic sensitivity: A first attempt with multi-speaker vowel data
3293                                    <br>
3294                                    <span class="w3-text w3-text-theme">
3295                                        Michael Lambropoulos, Frantz Clermont, Shunichi Ishihara
3296                                    </span>
3297                                </p>
3298                            </a>
3299                            <a class="w3-text" href="chung24_interspeech.html">
3300                                <p>
3301                                    Speaker-Independent Acoustic-to-Articulatory Inversion through Multi-Channel Attention Discriminator
3302                                    <br>
3303                                    <span class="w3-text w3-text-theme">
3304                                        Woo-Jin Chung, Hong-Goo Kang
3305                                    </span>
3306                                </p>
3307                            </a>
3308                            <a class="w3-text" href="weise24_interspeech.html">
3309                                <p>
3310                                    Speaker- and Text-Independent Estimation of Articulatory Movements and Phoneme Alignments from Speech
3311                                    <br>
3312                                    <span class="w3-text w3-text-theme">
3313                                        Tobias Weise, Philipp Klumpp, Kubilay Can Demir, Paula Andrea Pérez-Toro, Maria Schuster, Elmar Noeth, Bjoern Heismann, Andreas Maier, Seung Hee Yang
3314                                    </span>
3315                                </p>
3316                            </a>
3317                            <a class="w3-text" href="oura24_interspeech.html">
3318                                <p>
3319                                    Preprocessing for acoustic-to-articulatory inversion using real-time MRI movies of Japanese speech
3320                                    <br>
3321                                    <span class="w3-text w3-text-theme">
3322                                        Anna Oura, Hideaki Kikuchi, Tetsunori Kobayashi
3323                                    </span>
3324                                </p>
3325                            </a>
3326                        </div>
3327                    </div>
3328                    <br>
3329                    <div class="w3-content" style="height:10px"  id="Voice, Tones and F0"></div>
3330                    <div class="w3-card w3-round w3-white w3-padding">
3331                        <div class="w3-container"  style="margin-top:40px">
3332                            <h4 class="w3-center">Voice, Tones and F0</h4>
3333                            <hr>
3334                            <a class="w3-text" href="li24f_interspeech.html">
3335                                <p>
3336                                    Impact of the tonal factor on diphthong realizations in Standard Mandarin with Generalized Additive Mixed Models
3337                                    <br>
3338                                    <span class="w3-text w3-text-theme">
3339                                        Chenyu Li, Jalal Al-Tamimi
3340                                    </span>
3341                                </p>
3342                            </a>
3343                            <a class="w3-text" href="xiaowang24_interspeech.html">
3344                                <p>
3345                                    A Study on the Information Mechanism of the 3rd Tone Sandhi Rule in Mandarin Disyllabic Words
3346                                    <br>
3347                                    <span class="w3-text w3-text-theme">
3348                                        Liu Xiaowang, Jinsong Zhang
3349                                    </span>
3350                                </p>
3351                            </a>
3352                            <a class="w3-text" href="weirich24_interspeech.html">
3353                                <p>
3354                                    Gender and age based f0-variation in the German Plapper Corpus
3355                                    <br>
3356                                    <span class="w3-text w3-text-theme">
3357                                        Melanie Weirich, Daniel Duran, Stefanie Jannedy
3358                                    </span>
3359                                </p>
3360                            </a>
3361                            <a class="w3-text" href="xu24j_interspeech.html">
3362                                <p>
3363                                    Voice quality in telephone speech: Comparing acoustic measures between VoIP telephone and high-quality recordings
3364                                    <br>
3365                                    <span class="w3-text w3-text-theme">
3366                                        Chenzi Xu, Jessica Wormald, Paul Foulkes, Philip Harrison, Vincent Hughes, Poppy Welch, Finnian Kelly, David van der Vloed
3367                                    </span>
3368                                </p>
3369                            </a>
3370                            <a class="w3-text" href="gessinger24_interspeech.html">
3371                                <p>
3372                                    The Use of Modifiers and f0 in Remote Referential Communication with Human and Computer Partners
3373                                    <br>
3374                                    <span class="w3-text w3-text-theme">
3375                                        Iona Gessinger, Bistra Andreeva, Benjamin R. 
3375Cowan
3376                                    </span>
3377                                </p>
3378                            </a>
3379                        </div>
3380                    </div>
3381                    <br>
3382                    <div class="w3-content" style="height:10px"  id="Emotion Recognition: Resources and Benchmarks"></div>
3383                    <div class="w3-card w3-round w3-white w3-padding">
3384                        <div class="w3-container"  style="margin-top:40px">
3385                            <h4 class="w3-center">Emotion Recognition: Resources and Benchmarks</h4>
3386                            <hr>
3387                            <a class="w3-text" href="ma24b_interspeech.html">
3388                                <p>
3389                                    EmoBox: Multilingual Multi-corpus Speech Emotion Recognition Toolkit and Benchmark
3390                                    <br>
3391                                    <span class="w3-text w3-text-theme">
3392                                        Ziyang Ma, Mingjie Chen, Hezhao Zhang, Zhisheng Zheng, Wenxi Chen, Xiquan Li, Jiaxin Ye, Xie Chen, Thomas Hain
3393                                    </span>
3394                                </p>
3395                            </a>
3396                            <a class="w3-text" href="triantafyllopoulos24b_interspeech.html">
3397                                <p>
3398                                    INTERSPEECH 2009 Emotion Challenge Revisited: Benchmarking 15 Years of Progress in Speech Emotion Recognition
3399                                    <br>
3400                                    <span class="w3-text w3-text-theme">
3401                                        Andreas Triantafyllopoulos, Anton Batliner, Simon Rampp, Manuel Milling, Björn Schuller
3402                                    </span>
3403                                </p>
3404                            </a>
3405                            <a class="w3-text" href="ibrahim24_interspeech.html">
3406                                <p>
3407                                    What Does it Take to Generalize SER Model Across Datasets? A Comprehensive Benchmark
3408                                    <br>
3409                                    <span class="w3-text w3-text-theme">
3410                                        Adham Ibrahim, Shady Shehata, Ajinkya Kulkarni, Mukhtar Mohamed, Muhammad Abdul-Mageed
3411                                    </span>
3412                                </p>
3413                            </a>
3414                            <a class="w3-text" href="naini24_interspeech.html">
3415                                <p>
3416                                    WHiSER: White House Tapes Speech Emotion Recognition Corpus
3417                                    <br>
3418                                    <span class="w3-text w3-text-theme">
3419                                        Abinay Reddy Naini, Lucas Goncalves, Mary A. Kohler, Donita Robinson, Elizabeth Richerson, Carlos Busso
3420                                    </span>
3421                                </p>
3422                            </a>
3423                            <a class="w3-text" href="latif24_interspeech.html">
3424                                <p>
3425                                    Evaluating Transformer-Enhanced Deep Reinforcement Learning for Speech Emotion Recognition
3426                                    <br>
3427                                    <span class="w3-text w3-text-theme">
3428                                        Siddique Latif, Raja Jurdak, Björn W. Schuller
3429                                    </span>
3430                                </p>
3431                            </a>
3432                            <a class="w3-text" href="wang24ia_interspeech.html">
3433                                <p>
3434                                    Boosting Cross-Corpus Speech Emotion Recognition using CycleGAN with Contrastive Learning
3435                                    <br>
3436                                    <span class="w3-text w3-text-theme">
3437                                        Jincen Wang, Yan Zhao, Cheng Lu, Chuangao Tang, Sunan Li, Yuan Zong, Wenming Zheng
3438                                    </span>
3439                                </p>
3440                            </a>
3441                        </div>
3442                    </div>
3443                    <br>
3444                    <div class="w3-content" style="height:10px"  id="Speaker and Language Identification and Diarization"></div>
3445                    <div class="w3-card w3-round w3-white w3-padding">
3446                        <div class="w3-container"  style="margin-top:40px">
3447                            <h4 class="w3-center">Speaker and Language Identification and Diarization</h4>
3448                            <hr>
3449                            <a class="w3-text" href="rahou24_interspeech.html">
3450                                <p>
3451                                    Multi-latency look-ahead for streaming speaker segmentation
3452                                    <br>
3453                                    <span class="w3-text w3-text-theme">
3454                                        Bilal Rahou, Hervé Bredin
3455                                    </span>
3456                                </p>
3457                            </a>
3458                            <a class="w3-text" href="boeddeker24_interspeech.html">
3459                                <p>
3460                                    Once more Diarization: Improving meeting transcription systems through segment-level speaker reassignment
3461                                    <br>
3462                                    <span class="w3-text w3-text-theme">
3463                                        Christoph Boeddeker, Tobias Cord-Landwehr, Reinhold Haeb-Umbach
3464                                    </span>
3465                                </p>
3466                            </a>
3467                            <a class="w3-text" href="mariotte24_interspeech.html">
3468                                <p>
3469                                    ASoBO: Attentive Beamformer Selection for Distant Speaker Diarization in Meetings
3470                                    <br>
3471                                    <span class="w3-text w3-text-theme">
3472                                        Théo Mariotte, Anthony Larcher, Silvio Montrésor, Jean-Hugh Thomas
3473                                    </span>
3474                                </p>
3475                            </a>
3476                            <a class="w3-text" href="pirlogeanu24_interspeech.html">
3477                                <p>
3478                                    Hybrid-Diarization System with Overlap Post-Processing for the DISPLACE 2024 Challenge
3479                                    <br>
3480                                    <span class="w3-text w3-text-theme">
3481                                        Gabriel Pîrlogeanu, Octavian Pascu, Alexandru-Lucian Georgescu, Horia Cucu
3482                                    </span>
3483                                </p>
3484                            </a>
3485                            <a class="w3-text" href="kalluri24_interspeech.html">
3486                                <p>
3487                                    The Second DISPLACE Challenge: DIarization of SPeaker and LAnguage in Conversational Environments
3488                                    <br>
3489                                    <span class="w3-text w3-text-theme">
3490                                        Shareef Babu Kalluri, Prachi Singh, Pratik Roy Chowdhuri, Apoorva Kulkarni, Shikha Baghel, Pradyoth Hegde, Swapnil Sontakke, Deepak K T, S.R. Mahadeva Prasanna, Deepu Vijayasenan, Sriram Ganapathy
3491                                    </span>
3492                                </p>
3493                            </a>
3494                            <a class="w3-text" href="kalda24_interspeech.html">
3495                                <p>
3496                                    TalTech-IRIT-LIS Speaker and Language Diarization Systems for DISPLACE 2024
3497                                    <br>
3498                                    <span class="w3-text w3-text-theme">
3499                                        Joonas Kalda, Tanel Alumae, Martin Lebourdais, Hervé Bredin, Séverin Baroudi, Ricard Marxer
3500                                    </span>
3501                                </p>
3502                            </a>
3503                            <a class="w3-text" href="hao24b_interspeech.html">
3504                                <p>
3505                                    Exploring Energy-Based Models for Out-of-Distribution Detection in Dialect Identification
3506                                    <br>
3507                                    <span class="w3-text w3-text-theme">
3508                                        Yaqian Hao, Chenguang Hu, Yingying Gao, Shilei Zhang, Junlan Feng
3509                                    </span>
3510                                </p>
3511                            </a>
3512                            <a class="w3-text" href="valente24_interspeech.html">
3513                                <p>
3514                                    Exploring Spoken Language Identification Strategies for Automatic Transcription of Multilingual Broadcast and Institutional Speech
3515                                    <br>
3516                                    <span class="w3-text w3-text-theme">
3517                                        Martina Valente, Fabio Brugnara, Giovanni Morrone, Enrico Zovato, Leonardo Badino
3518                                    </span>
3519                                </p>
3520                            </a>
3521                            <a class="w3-text" href="paturi24_interspeech.html">
3522                                <p>
3523                                    AG-LSEC: Audio Grounded Lexical Speaker Error Correction
3524                                    <br>
3525                                    <span class="w3-text w3-text-theme">
3526                                        Rohit Paturi, Xiang Li, Sundararajan Srinivasan
3527                                    </span>
3528                                </p>
3529                            </a>
3530                            <a class="w3-text" href="su24_interspeech.html">
3531                                <p>
3532                                    Speaker Change Detection with Weighted-sum Knowledge Distillation based on Self-supervised Pre-trained Models
3533                                    <br>
3534                                    <span class="w3-text w3-text-theme">
3535                                        Hang Su, Yuxiang Kong, Lichun Fan, Peng Gao, Yujun Wang, Zhiyong Wu
3536                                    </span>
3537                                </p>
3538                            </a>
3539                            <a class="w3-text" href="makishima24_interspeech.html">
3540                                <p>
3541                                    SOMSRED: Sequential Output Modeling for Joint Multi-talker Overlapped Speech Recognition and Speaker Diarization
3542                                    <br>
3543                                    <span class="w3-text w3-text-theme">
3544                                        Naoki Makishima, Naotaka Kawata, Mana Ihori, Tomohiro Tanaka, Shota Orihashi, Atsushi Ando, Ryo Masumura
3545                                    </span>
3546                                </p>
3547                            </a>
3548                            <a class="w3-text" href="munakata24_interspeech.html">
3549                                <p>
3550                                    Song Data Cleansing for End-to-End Neural Singer Diarization Using Neural Analysis and Synthesis Framework
3551                                    <br>
3552                                    <span class="w3-text w3-text-theme">
3553                                        Hokuto Munakata, Ryo Terashima, Yusuke Fujita
3554                                    </span>
3555                                </p>
3556                            </a>
3557                        </div>
3558                    </div>
3559                    <br>
3560                    <div class="w3-content" style="height:10px"  id="Audio-Text Retrieval"></div>
3561                    <div class="w3-card w3-round w3-white w3-padding">
3562                        <div class="w3-container"  style="margin-top:40px">
3563                            <h4 class="w3-center">Audio-Text Retrieval</h4>
3564                            <hr>
3565                            <a class="w3-text" href="xin24_interspeech.html">
3566                                <p>
3567                                    DiffATR: Diffusion-based Generative Modeling for Audio-Text Retrieval
3568                                    <br>
3569                                    <span class="w3-text w3-text-theme">
3570                                        Yifei Xin, Xuxin Cheng, Zhihong Zhu, Xusheng Yang, Yuexian Zou
3571                                    </span>
3572                                </p>
3573                            </a>
3574                            <a class="w3-text" href="yan24_interspeech.html">
3575                                <p>
3576                                    Bridging Language Gaps in Audio-Text Retrieval
3577                                    <br>
3578                                    <span class="w3-text w3-text-theme">
3579                                        Zhiyong Yan, Heinrich Dinkel, Yongqing Wang, Jizhong Liu, Junbo Zhang, Yujun Wang, Bin Wang
3580                                    </span>
3581                                </p>
3582                            </a>
3583                            <a class="w3-text" href="deshmukh24_interspeech.html">
3584                                <p>
3585                                    Domain Adaptation for Contrastive Audio-Language Models
3586                                    <br>
3587                                    <span class="w3-text w3-text-theme">
3588                                        Soham Deshmukh, Rita Singh, Bhiksha Raj
3589                                    </span>
3590                                </p>
3591                            </a>
3592                            <a class="w3-text" href="paissan24_interspeech.html">
3593                                <p>
3594                                    tinyCLAP: Distilling Constrastive Language-Audio Pretrained Models
3595                                    <br>
3596                                    <span class="w3-text w3-text-theme">
3597                                        Francesco Paissan, Elisabetta Farella
3598                                    </span>
3599                                </p>
3600                            </a>
3601                            <a class="w3-text" href="kim24f_interspeech.html">
3602                                <p>
3603                                    BTS: Bridging Text and Sound Modalities for Metadata-Aided Respiratory Sound Classification
3604                                    <br>
3605                                    <span class="w3-text w3-text-theme">
3606                                        June-Woo Kim, Miika Toikkanen, Yera Choi, Seoung-Eun Moon, Ho-Young Jung
3607                                    </span>
3608                                </p>
3609                            </a>
3610                            <a class="w3-text" href="tang24b_interspeech.html">
3611                                <p>
3612                                    Enhanced Feature Learning with Normalized Knowledge Distillation for Audio Tagging
3613                                    <br>
3614                                    <span class="w3-text w3-text-theme">
3615                                        Yuwu Tang, Ziang Ma, Haitao Zhang
3616                                    </span>
3617                                </p>
3618                            </a>
3619                        </div>
3620                    </div>
3621                    <br>
3622                    <div class="w3-content" style="height:10px"  id="Speech Enhancement"></div>
3623                    <div class="w3-card w3-round w3-white w3-padding">
3624                        <div class="w3-container"  style="margin-top:40px">
3625                            <h4 class="w3-center">Speech Enhancement</h4>
3626                            <hr>
3627                            <a class="w3-text" href="liu24n_interspeech.html">
3628                                <p>
3629                                    RaD-Net 2: A causal two-stage repairing and denoising speech enhancement network with knowledge distillation and complex axial self-attention
3630                                    <br>
3631                                    <span class="w3-text w3-text-theme">
3632                                        Mingshuai Liu, Zhuangqi Chen, Xiaopeng Yan, Yuanjun Lv, Xianjun Xia, Chuanzeng Huang, Yijian Xiao, Lei Xie
3633                                    </span>
3634                                </p>
3635                            </a>
3636                            <a class="w3-text" href="liu24o_interspeech.html">
3637                                <p>
3638                                    DNN-based monaural speech enhancement using alternate analysis windows for phase and magnitude modification
3639                                    <br>
3640                                    <span class="w3-text w3-text-theme">
3641                                        Xi Liu, John H.L. Hansen
3642                                    </span>
3643                                </p>
3644                            </a>
3645                            <a class="w3-text" href="li24aa_interspeech.html">
3646                                <p>
3647                                    Improved Remixing Process for Domain Adaptation-Based Speech Enhancement by Mitigating Data Imbalance in Signal-to-Noise Ratio
3648                                    <br>
3649                                    <span class="w3-text w3-text-theme">
3650                                        Li Li, Shogo Seki
3651                                    </span>
3652                                </p>
3653                            </a>
3654                            <a class="w3-text" href="zhang24_interspeech.html">
3655                                <p>
3656                                    Neural Network Augmented Kalman Filter for Robust Acoustic Howling Suppression
3657                                    <br>
3658                                    <span class="w3-text w3-text-theme">
3659                                        Yixuan Zhang, Hao Zhang, Meng Yu, Dong Yu
3660                                    </span>
3661                                </p>
3662                            </a>
3663                            <a class="w3-text" href="li24w_interspeech.html">
3664                                <p>
3665                                    Improving Speech Enhancement by Integrating Inter-Channel and Band Features with Dual-branch Conformer
3666                                    <br>
3667                                    <span class="w3-text w3-text-theme">
3668                                        Jizhen Li, Xinmeng Xu, Weiping Tu, Yuhong Yang, Rong Zhu
3669                                    </span>
3670                                </p>
3671                            </a>
3672                            <a class="w3-text" href="zhang24n_interspeech.html">
3673                                <p>
3674                                    An Exploration of Length Generalization in Transformer-Based Speech Enhancement
3675                                    <br>
3676                                    <span class="w3-text w3-text-theme">
3677                                        Qiquan Zhang, Hongxu Zhu, Xinyuan Qian, Eliathamby Ambikairajah, Haizhou Li
3678                                    </span>
3679                                </p>
3680                            </a>
3681                            <a class="w3-text" href="guan24_interspeech.html">
3682                                <p>
3683                                    Reducing Speech Distortion and Artifacts for Speech Enhancement by Loss Function
3684                                    <br>
3685                                    <span class="w3-text w3-text-theme">
3686                                        Haixin Guan, Wei Dai, Guangyong Wang, Xiaobin Tan, Peng Li, Jiaen Liang
3687                                    </span>
3688                                </p>
3689                            </a>
3690                            <a class="w3-text" href="mawalim24_interspeech.html">
3691                                <p>
3692                                    Are Recent Deep Learning-Based Speech Enhancement Methods Ready to Confront Real-World Noisy Environments?
3693                                    <br>
3694                                    <span class="w3-text w3-text-theme">
3695                                        Candy Olivia Mawalim, Shogo Okada, Masashi Unoki
3696                                    </span>
3697                                </p>
3698                            </a>
3699                            <a class="w3-text" href="zhang24i_interspeech.html">
3700                                <p>
3701                                    Beyond Performance Plateaus: A Comprehensive Study on Scalability in Speech Enhancement
3702                                    <br>
3703                                    <span class="w3-text w3-text-theme">
3704                                        Wangyou Zhang, Kohei Saijo, Jee-weon Jung, Chenda Li, Shinji Watanabe, Yanmin Qian
3705                                    </span>
3706                                </p>
3707                            </a>
3708                        </div>
3709                    </div>
3710                    <br>
3711                    <div class="w3-content" style="height:10px"  id="Speech Coding"></div>
3712                    <div class="w3-card w3-round w3-white w3-padding">
3713                        <div class="w3-container"  style="margin-top:40px">
3714                            <h4 class="w3-center">Speech Coding</h4>
3715                            <hr>
3716                            <a class="w3-text" href="zhang24g_interspeech.html">
3717                                <p>
3718                                    TD-PLC: A Semantic-Aware Speech Encoding for Improved Packet Loss Concealment
3719                                    <br>
3720                                    <span class="w3-text w3-text-theme">
3721                                        Jinghong Zhang, Zugang Zhao, Yonghui Liu, Jianbing Liu, Zhiqiang He, Kai Niu
3722                                    </span>
3723                                </p>
3724                            </a>
3725                            <a class="w3-text" href="zhang24m_interspeech.html">
3726                                <p>
3727                                    BS-PLCNet 2: Two-stage Band-split Packet Loss Concealment Network with Intra-model Knowledge Distillation
3728                                    <br>
3729                                    <span class="w3-text w3-text-theme">
3730                                        Zihan Zhang, Xianjun Xia, Chuanzeng Huang, Yijian Xiao, Lei Xie
3731                                    </span>
3732                                </p>
3733                            </a>
3734                            <a class="w3-text" href="gupta24c_interspeech.html">
3735                                <p>
3736                                    On Improving Error Resilience of Neural End-to-End Speech Coders
3737                                    <br>
3738                                    <span class="w3-text w3-text-theme">
3739                                        Kishan Gupta, Nicola Pia, Srikanth Korse, Andreas Brendel, Guillaume Fuchs, Markus Multrus
3740                                    </span>
3741                                </p>
3742                            </a>
3743                            <a class="w3-text" href="muller24c_interspeech.html">
3744                                <p>
3745                                    Speech quality evaluation of neural audio codecs
3746                                    <br>
3747                                    <span class="w3-text w3-text-theme">
3748                                        Thomas Muller, Stephane Ragot, Laetitia Gros, Pierrick Philippe, Pascal Scalart
3749                                    </span>
3750                                </p>
3751                            </a>
3752                            <a class="w3-text" href="ai24b_interspeech.html">
3753                                <p>
3754                                    A Low-Bitrate Neural Audio Codec Framework with Bandwidth Reduction and Recovery for High-Sampling-Rate Waveforms
3755                                    <br>
3756                                    <span class="w3-text w3-text-theme">
3757                                        Yang Ai, Ye-Xin Lu, Xiao-Hang Jiang, Zheng-Yan Sheng, Rui-Chen Zheng, Zhen-Hua Ling
3758                                    </span>
3759                                </p>
3760                            </a>
3761                            <a class="w3-text" href="wu24p_interspeech.html">
3762                                <p>
3763                                    CodecFake: Enhancing Anti-Spoofing Models Against Deepfake Audios from Codec-Based Speech Synthesis Systems
3764                                    <br>
3765                                    <span class="w3-text w3-text-theme">
3766                                        Haibin Wu, Yuan Tseng, Hung-yi Lee
3767                                    </span>
3768                                </p>
3769                            </a>
3770                        </div>
3771                    </div>
3772                    <br>
3773                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Expressivity and Emotion"></div>
3774                    <div class="w3-card w3-round w3-white w3-padding">
3775                        <div class="w3-container"  style="margin-top:40px">
3776                            <h4 class="w3-center">Speech Synthesis: Expressivity and Emotion</h4>
3777                            <hr>
3778                            <a class="w3-text" href="li24pa_interspeech.html">
3779                                <p>
3780                                    GTR-Voice: Articulatory Phonetics Informed Controllable Expressive Speech Synthesis
3781                                    <br>
3782                                    <span class="w3-text w3-text-theme">
3783                                        Zehua Kcriss Li, Meiying Melissa Chen, Yi Zhong, Pinxin Liu, Zhiyao Duan
3784                                    </span>
3785                                </p>
3786                            </a>
3787                            <a class="w3-text" href="seong24b_interspeech.html">
3788                                <p>
3789                                    TSP-TTS: Text-based Style Predictor with Residual Vector Quantization for Expressive Text-to-Speech
3790                                    <br>
3791                                    <span class="w3-text w3-text-theme">
3792                                        Donghyun Seong, Hoyoung Lee, Joon-Hyuk Chang
3793                                    </span>
3794                                </p>
3795                            </a>
3796                            <a class="w3-text" href="li24na_interspeech.html">
3797                                <p>
3798                                    Spontaneous Style Text-to-Speech Synthesis with Controllable Spontaneous Behaviors Based on Language Models
3799                                    <br>
3800                                    <span class="w3-text w3-text-theme">
3801                                        Weiqin Li, Peiji Yang, Yicheng Zhong, Yixuan Zhou, Zhisheng Wang, Zhiyong Wu, Xixin Wu, Helen Meng
3802                                    </span>
3803                                </p>
3804                            </a>
3805                            <a class="w3-text" href="guo24d_interspeech.html">
3806                                <p>
3807                                    Text-aware and Context-aware Expressive Audiobook Speech Synthesis
3808                                    <br>
3809                                    <span class="w3-text w3-text-theme">
3810                                        Dake Guo, Xinfa Zhu, Liumeng Xue, Yongmao Zhang, Wenjie Tian, Lei Xie
3811                                    </span>
3812                                </p>
3813                            </a>
3814                            <a class="w3-text" href="bott24_interspeech.html">
3815                                <p>
3816                                    Controlling Emotion in Text-to-Speech with Natural Language Prompts
3817                                    <br>
3818                                    <span class="w3-text w3-text-theme">
3819                                        Thomas Bott, Florian Lux, Ngoc Thang Vu
3820                                    </span>
3821                                </p>
3822                            </a>
3823                            <a class="w3-text" href="xue24b_interspeech.html">
3824                                <p>
3825                                    Retrieval Augmented Generation in Prompt-based Text-to-Speech Synthesis with Context-Aware Contrastive Language-Audio Pretraining
3826                                    <br>
3827                                    <span class="w3-text w3-text-theme">
3828                                        Jinlong Xue, Yayue Deng, Yingming Gao, Ya Li
3829                                    </span>
3830                                </p>
3831                            </a>
3832                            <a class="w3-text" href="kalyan24_interspeech.html">
3833                                <p>
3834                                    Emotion Arithmetic: Emotional Speech Synthesis via Weight Space Interpolation
3835                                    <br>
3836                                    <span class="w3-text w3-text-theme">
3837                                        Pavan Kalyan, Preeti Rao, Preethi Jyothi, Pushpak Bhattacharyya
3838                                    </span>
3839                                </p>
3840                            </a>
3841                            <a class="w3-text" href="cho24_interspeech.html">
3842                                <p>
3843                                    EmoSphere-TTS: Emotional Style and Intensity Modeling via Spherical Emotion Vector for Controllable Emotional Text-to-Speech
3844                                    <br>
3845                                    <span class="w3-text w3-text-theme">
3846                                        Deok-Hyeon Cho, Hyung-Seok Oh, Seung-Bin Kim, Sang-Hoon Lee, Seong-Whan Lee
3847                                    </span>
3848                                </p>
3849                            </a>
3850                            <a class="w3-text" href="li24da_interspeech.html">
3851                                <p>
3852                                    Expressive paragraph text-to-speech synthesis with multi-step variational autoencoder
3853                                    <br>
3854                                    <span class="w3-text w3-text-theme">
3855                                        Xuyuan Li, Zengqiang Shang, Peiyang Shi, Hua Hua, Ta Li, Pengyuan Zhang
3856                                    </span>
3857                                </p>
3858                            </a>
3859                            <a class="w3-text" href="yu24b_interspeech.html">
3860                                <p>
3861                                    Differentiable Time-Varying Linear Prediction in the Context of End-to-End Analysis-by-Synthesis
3862                                    <br>
3863                                    <span class="w3-text w3-text-theme">
3864                                        Chin-Yun Yu, György Fazekas
3865                                    </span>
3866                                </p>
3867                            </a>
3868                        </div>
3869                    </div>
3870                    <br>
3871                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Tools and Data"></div>
3872                    <div class="w3-card w3-round w3-white w3-padding">
3873                        <div class="w3-container"  style="margin-top:40px">
3874                            <h4 class="w3-center">Speech Synthesis: Tools and Data</h4>
3875                            <hr>
3876                            <a class="w3-text" href="saito24_interspeech.html">
3877                                <p>
3878                                    SRC4VC: Smartphone-Recorded Corpus for Voice Conversion Benchmark
3879                                    <br>
3880                                    <span class="w3-text w3-text-theme">
3881                                        Yuki Saito, Takuto Igarashi, Kentaro Seki, Shinnosuke Takamichi, Ryuichi Yamamoto, Kentaro Tachibana, Hiroshi Saruwatari
3882                                    </span>
3883                                </p>
3884                            </a>
3885                            <a class="w3-text" href="srinivasavaradhan24_interspeech.html">
3886                                <p>
3887                                    Rasa: Building Expressive Speech Synthesis Systems for Indian Languages in Low-resource Settings
3888                                    <br>
3889                                    <span class="w3-text w3-text-theme">
3890                                        Praveen Srinivasa Varadhan, Ashwin Sankar, Giri Raju, Mitesh M Khapra
3891                                    </span>
3892                                </p>
3893                            </a>
3894                            <a class="w3-text" href="ma24c_interspeech.html">
3895                                <p>
3896                                    FLEURS-R: A Restored Multilingual Speech Corpus for Generation Tasks
3897                                    <br>
3898                                    <span class="w3-text w3-text-theme">
3899                                        Min Ma, Yuma Koizumi, Shigeki Karita, Heiga Zen, Jason Riesa, Haruko Ishikawa, Michiel Bacchiani
3900                                    </span>
3901                                </p>
3902                            </a>
3903                            <a class="w3-text" href="ma24d_interspeech.html">
3904                                <p>
3905                                    WenetSpeech4TTS: A 12,800-hour Mandarin TTS Corpus for Large Speech Generation Model Benchmark
3906                                    <br>
3907                                    <span class="w3-text w3-text-theme">
3908                                        Linhan Ma, Dake Guo, Kun Song, Yuepeng Jiang, Shuai Wang, Liumeng Xue, Weiming Xu, Huan Zhao, Binbin Zhang, Lei Xie
3909                                    </span>
3910                                </p>
3911                            </a>
3912                            <a class="w3-text" href="yang24d_interspeech.html">
3913                                <p>
3914                                    MSceneSpeech: A Multi-Scene Speech Dataset For Expressive Speech Synthesis
3915                                    <br>
3916                                    <span class="w3-text w3-text-theme">
3917                                        Qian Yang, Jialong Zuo, Zhe Su, Ziyue Jiang, Mingze Li, Zhou Zhao, Feiyang Chen, Zhefeng Wang, Baoxing Huai
3918                                    </span>
3919                                </p>
3920                            </a>
3921                            <a class="w3-text" href="kawamura24_interspeech.html">
3922                                <p>
3923                                    LibriTTS-P: A Corpus with Speaking Style and Speaker Identity Prompts for Text-to-Speech and Style Captioning
3924                                    <br>
3925                                    <span class="w3-text w3-text-theme">
3926                                        Masaya Kawamura, Ryuichi Yamamoto, Yuma Shirahata, Takuya Hasumi, Kentaro Tachibana
3927                                    </span>
3928                                </p>
3929                            </a>
3930                            <a class="w3-text" href="ogun24_interspeech.html">
3931                                <p>
3932                                    1000 African Voices: Advancing inclusive multi-speaker multi-accent speech synthesis
3933                                    <br>
3934                                    <span class="w3-text w3-text-theme">
3935                                        Sewade Ogun, Abraham T. Owodunni, Tobi Olatunji, Eniola Alese, Babatunde Oladimeji, Tejumade Afonja, Kayode Olaleye, Naome A. Etori, Tosin Adewumi
3936                                    </span>
3937                                </p>
3938                            </a>
3939                            <a class="w3-text" href="take24_interspeech.html">
3940                                <p>
3941                                    SaSLaW: Dialogue Speech Corpus with Audio-visual Egocentric Information Toward Environment-adaptive Dialogue Speech Synthesis
3942                                    <br>
3943                                    <span class="w3-text w3-text-theme">
3944                                        Osamu Take, Shinnosuke Takamichi, Kentaro Seki, Yoshiaki Bando, Hiroshi Saruwatari
3945                                    </span>
3946                                </p>
3947                            </a>
3948                        </div>
3949                    </div>
3950                    <br>
3951                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Singing Voice Synthesis"></div>
3952                    <div class="w3-card w3-round w3-white w3-padding">
3953                        <div class="w3-container"  style="margin-top:40px">
3954                            <h4 class="w3-center">Speech Synthesis: Singing Voice Synthesis</h4>
3955                            <hr>
3956                            <a class="w3-text" href="kim24i_interspeech.html">
3957                                <p>
3958                                    MakeSinger: A Semi-Supervised Training Method for Data-Efficient Singing Voice Synthesis via Classifier-free Diffusion Guidance
3959                                    <br>
3960                                    <span class="w3-text w3-text-theme">
3961                                        Semin Kim, Myeonghun Jeong, Hyeonseung Lee, Minchan Kim, Byoung Jin Choi, Nam Soo Kim
3962                                    </span>
3963                                </p>
3964                            </a>
3965                            <a class="w3-text" href="okamoto24_interspeech.html">
3966                                <p>
3967                                    Challenge of Singing Voice Synthesis Using Only Text-To-Speech Corpus With FIRNet Source-Filter Neural Vocoder
3968                                    <br>
3969                                    <span class="w3-text w3-text-theme">
3970                                        Takuma Okamoto, Yamato Ohtani, Sota Shimizu, Tomoki Toda, Hisashi Kawai
3971                                    </span>
3972                                </p>
3973                            </a>
3974                            <a class="w3-text" href="kim24p_interspeech.html">
3975                                <p>
3976                                    Period Singer: Integrating Periodic and Aperiodic Variational Autoencoders for Natural-Sounding End-to-End Singing Voice Synthesis
3977                                    <br>
3978                                    <span class="w3-text w3-text-theme">
3979                                        Taewoo Kim, Choonsang Cho, Young Han Lee
3980                                    </span>
3981                                </p>
3982                            </a>
3983                            <a class="w3-text" href="shi24_interspeech.html">
3984                                <p>
3985                                    Singing Voice Data Scaling-up: An Introduction to ACE-Opencpop and ACE-KiSing
3986                                    <br>
3987                                    <span class="w3-text w3-text-theme">
3988                                        Jiatong Shi, Yueqian Lin, Xinyi Bai, Keyi Zhang, Yuning Wu, Yuxun Tang, Yifeng Yu, Qin Jin, Shinji Watanabe
3989                                    </span>
3990                                </p>
3991                            </a>
3992                            <a class="w3-text" href="hwang24_interspeech.html">
3993                                <p>
3994                                    X-Singer: Code-Mixed Singing Voice Synthesis via Cross-Lingual Learning
3995                                    <br>
3996                                    <span class="w3-text w3-text-theme">
3997                                        Ji-Sang Hwang, Hyeongrae Noh, Yoonseok Hong, Insoo Oh
3998                                    </span>
3999                                </p>
4000                            </a>
4001                            <a class="w3-text" href="gao24e_interspeech.html">
4002                                <p>
4003                                    An End-to-End Approach for Chord-Conditioned Song Generation
4004                                    <br>
4005                                    <span class="w3-text w3-text-theme">
4006                                        Shuochen Gao, Shun Lei, Fan Zhuo, Hangyu Liu, Feng Liu, Boshi Tang, Qiaochu Huang, Shiyin Kang, Zhiyong Wu
4007                                    </span>
4008                                </p>
4009                            </a>
4010                        </div>
4011                    </div>
4012                    <br>
4013                    <div class="w3-content" style="height:10px"  id="LLM in ASR"></div>
4014                    <div class="w3-card w3-round w3-white w3-padding">
4015                        <div class="w3-container"  style="margin-top:40px">
4016                            <h4 class="w3-center">LLM in ASR</h4>
4017                            <hr>
4018                            <a class="w3-text" href="baskar24_interspeech.html">
4019                                <p>
4020                                    Speech Prefix-Tuning with RNNT Loss for Improving LLM Predictions
4021                                    <br>
4022                                    <span class="w3-text w3-text-theme">
4023                                        Murali Karthick Baskar, Andrew Rosenberg, Bhuvana Ramabhadran, Neeraj Gaur, Zhong Meng
4024                                    </span>
4025                                </p>
4026                            </a>
4027                            <a class="w3-text" href="seide24_interspeech.html">
4028                                <p>
4029                                    Speech ReaLLM – Real-time Speech Recognition with Multimodal Language Models by Teaching the Flow of Time
4030                                    <br>
4031                                    <span class="w3-text w3-text-theme">
4032                                        Frank Seide, Yangyang Shi, Morrie Doulaty, Yashesh Gaur, Junteng Jia, Chunyang Wu
4033                                    </span>
4034                                </p>
4035                            </a>
4036                            <a class="w3-text" href="li24t_interspeech.html">
4037                                <p>
4038                                    A Transcription Prompt-based Efficient Audio Large Language Model for Robust Speech Recognition
4039                                    <br>
4040                                    <span class="w3-text w3-text-theme">
4041                                        Yangze Li, Xiong Wang, Songjun Cao, Yike Zhang, Long Ma, Lei Xie
4042                                    </span>
4043                                </p>
4044                            </a>
4045                            <a class="w3-text" href="tang24_interspeech.html">
4046                                <p>
4047                                    Pinyin Regularization in Error Correction for Chinese Speech Recognition with Large Language Models
4048                                    <br>
4049                                    <span class="w3-text w3-text-theme">
4050                                        Zhiyuan Tang, Dong Wang, Shen Huang, Shidong Shang
4051                                    </span>
4052                                </p>
4053                            </a>
4054                        </div>
4055                    </div>
4056                    <br>
4057                    <div class="w3-content" style="height:10px"  id="Vision and Speech"></div>
4058                    <div class="w3-card w3-round w3-white w3-padding">
4059                        <div class="w3-container"  style="margin-top:40px">
4060                            <h4 class="w3-center">Vision and Speech</h4>
4061                            <hr>
4062                            <a class="w3-text" href="kim24g_interspeech.html">
4063                                <p>
4064                                    AVCap: Leveraging Audio-Visual Features as Text Tokens for Captioning
4065                                    <br>
4066                                    <span class="w3-text w3-text-theme">
4067                                        Jongsuk Kim, Jiwon Shin, Junmo Kim
4068                                    </span>
4069                                </p>
4070                            </a>
4071                            <a class="w3-text" href="ghosh24b_interspeech.html">
4072                                <p>
4073                                    LipGER: Visually-Conditioned Generative Error Correction for Robust Automatic Speech Recognition
4074                                    <br>
4075                                    <span class="w3-text w3-text-theme">
4076                                        Sreyan Ghosh, Sonal Kumar, Ashish Seth, Purva Chiniya, Utkarsh Tyagi, Ramani Duraiswami, Dinesh Manocha
4077                                    </span>
4078                                </p>
4079                            </a>
4080                            <a class="w3-text" href="li24v_interspeech.html">
4081                                <p>
4082                                    Joint Speaker Features Learning for Audio-visual Multichannel Speech Separation and Recognition
4083                                    <br>
4084                                    <span class="w3-text w3-text-theme">
4085                                        Guinan Li, Jiajun Deng, Youjun Chen, Mengzhe Geng, Shujie Hu, Zhe Li, Zengrui Jin, Tianzi Wang, Xurong Xie, Helen Meng, Xunying Liu
4086                                    </span>
4087                                </p>
4088                            </a>
4089                            <a class="w3-text" href="chen24y_interspeech.html">
4090                                <p>
4091                                    CNVSRC 2023: The First Chinese Continuous Visual Speech Recognition Challenge
4092                                    <br>
4093                                    <span class="w3-text w3-text-theme">
4094                                        Chen Chen, Zehua Liu, Xiaolou Li, Lantian Li, Dong Wang
4095                                    </span>
4096                                </p>
4097                            </a>
4098                        </div>
4099                    </div>
4100                    <br>
4101                    <div class="w3-content" style="height:10px"  id="Spoken Document Summarization"></div>
4102                    <div class="w3-card w3-round w3-white w3-padding">
4103                        <div class="w3-container"  style="margin-top:40px">
4104                            <h4 class="w3-center">Spoken Document Summarization</h4>
4105                            <hr>
4106                            <a class="w3-text" href="kroll24_interspeech.html">
4107                                <p>
4108                                    Optimizing the role of human evaluation in LLM-based spoken document summarization systems
4109                                    <br>
4110                                    <span class="w3-text w3-text-theme">
4111                                        Margaret Kroll, Kelsey Kraus
4112                                    </span>
4113                                </p>
4114                            </a>
4115                            <a class="w3-text" href="ryu24_interspeech.html">
4116                                <p>
4117                                    Key-Element-Informed sLLM Tuning for Document Summarization
4118                                    <br>
4119                                    <span class="w3-text w3-text-theme">
4120                                        Sangwon Ryu, Heejin Do, Yunsu Kim, Gary Geunbae Lee, Jungseul Ok
4121                                    </span>
4122                                </p>
4123                            </a>
4124                            <a class="w3-text" href="matsuura24_interspeech.html">
4125                                <p>
4126                                    Sentence-wise Speech Summarization: Task, Datasets, and End-to-End Modeling with LM Knowledge Distillation
4127                                    <br>
4128                                    <span class="w3-text w3-text-theme">
4129                                        Kohei Matsuura, Takanori Ashihara, Takafumi Moriya, Masato Mimura, Takatomo Kano, Atsunori Ogawa, Marc Delcroix
4130                                    </span>
4131                                </p>
4132                            </a>
4133                            <a class="w3-text" href="shang24_interspeech.html">
4134                                <p>
4135                                    An End-to-End Speech Summarization Using Large Language Model
4136                                    <br>
4137                                    <span class="w3-text w3-text-theme">
4138                                        Hengchao Shang, Zongyao Li, Jiaxin Guo, Shaojun Li, Zhiqiang Rao, Yuanchang Luo, Daimeng Wei, Hao Yang
4139                                    </span>
4140                                </p>
4141                            </a>
4142                            <a class="w3-text" href="kang24d_interspeech.html">
4143                                <p>
4144                                    Prompting Large Language Models with Audio for General-Purpose Speech Summarization
4145                                    <br>
4146                                    <span class="w3-text w3-text-theme">
4147                                        Wonjune Kang, Deb Roy
4148                                    </span>
4149                                </p>
4150                            </a>
4151                            <a class="w3-text" href="leduc24_interspeech.html">
4152                                <p>
4153                                    Real-time Speech Summarization for Medical Conversations
4154                                    <br>
4155                                    <span class="w3-text w3-text-theme">
4156                                        Khai Le-Duc, Khai-Nguyen Nguyen, Long Vo-Dang, Truong-Son Hy
4157                                    </span>
4158                                </p>
4159                            </a>
4160                        </div>
4161                    </div>
4162                    <br>
4163                    <div class="w3-content" style="height:10px"  id="Speech and Language in Health: from Remote Monitoring to Medical Conversations - 2 (Special Sessions)"></div>
4164                    <div class="w3-card w3-round w3-white w3-padding">
4165                        <div class="w3-container"  style="margin-top:40px">
4166                            <h4 class="w3-center">Speech and Language in Health: from Remote Monitoring to Medical Conversations - 2 (Special Sessions)</h4>
4167                            <hr>
4168                            <a class="w3-text" href="escobargrisales24_interspeech.html">
4169                                <p>
4170                                    It’s Time to Take Action: Acoustic Modeling of Motor Verbs to Detect Parkinson’s Disease
4171                                    <br>
4172                                    <span class="w3-text w3-text-theme">
4173                                        Daniel Escobar-Grisales, Cristian David Ríos-Urrego, Ilja Baumann, Korbinian Riedhammer, Elmar Noeth, Tobias Bocklet, Adolfo M. Garcia, Juan Rafael Orozco-Arroyave
4174                                    </span>
4175                                </p>
4176                            </a>
4177                            <a class="w3-text" href="maisonneuve24_interspeech.html">
4178                                <p>
4179                                    Towards objective and interpretable speech disorder assessment: a comparative analysis of CNN and transformer-based models
4180                                    <br>
4181                                    <span class="w3-text w3-text-theme">
4182                                        Malo Maisonneuve, Corinne Fredouille, Muriel Lalain, Alain Ghio, Virginie Woisard
4183                                    </span>
4184                                </p>
4185                            </a>
4186                            <a class="w3-text" href="botelho24_interspeech.html">
4187                                <p>
4188                                    Macro-descriptors for Alzheimer's disease detection using large language models
4189                                    <br>
4190                                    <span class="w3-text w3-text-theme">
4191                                        Catarina Botelho, John Mendonça, Anna Pompili, Tanja Schultz, Alberto Abad, Isabel Trancoso
4192                                    </span>
4193                                </p>
4194                            </a>
4195                            <a class="w3-text" href="braun24_interspeech.html">
4196                                <p>
4197                                    Infusing Acoustic Pause Context into Text-Based Dementia Assessment
4198                                    <br>
4199                                    <span class="w3-text w3-text-theme">
4200                                        Franziska Braun, Sebastian P. Bayerl, Florian Hönig, Hartmut Lehfeld, Thomas Hillemacher, Tobias Bocklet, Korbinian Riedhammer
4201                                    </span>
4202                                </p>
4203                            </a>
4204                            <a class="w3-text" href="roesler24_interspeech.html">
4205                                <p>
4206                                    Towards Scalable Remote Assessment of Mild Cognitive Impairment Via Multimodal Dialog
4207                                    <br>
4208                                    <span class="w3-text w3-text-theme">
4209                                        Oliver Roesler, Jackson Liscombe, Michael Neumann, Hardik Kothare, Abhishek Hosamath, Lakshmi Arbatti, Doug Habberstad, Christiane Suendermann-Oeft, Meredith Bartlett, Cathy Zhang, Nikhil Sukhdev, Kolja Wilms, Anusha Badathala, Sandrine Istas, Steve Ruhmel, Bryan Hansen, Madeline Hannan, David Henley, Arthur Wallace, Ira Shoulson, David Suendermann-Oeft, Vikram Ramanarayanan
4210                                    </span>
4211                                </p>
4212                            </a>
4213                            <a class="w3-text" href="barberis24_interspeech.html">
4214                                <p>
4215                                    Automatic recognition and detection of aphasic natural speech
4216                                    <br>
4217                                    <span class="w3-text w3-text-theme">
4218                                        Mara Barberis, Pieter De Clercq, Bastiaan Tamm, Hugo Van hamme, Maaike Vandermosten
4219                                    </span>
4220                                </p>
4221                            </a>
4222                            <a class="w3-text" href="sanguedolce24_interspeech.html">
4223                                <p>
4224                                    When Whisper Listens to Aphasia: Advancing Robust Post-Stroke Speech Recognition
4225                                    <br>
4226                                    <span class="w3-text w3-text-theme">
4227                                        Giulia Sanguedolce, Sophie Brook, Dragos C. Gruia, Patrick A. Naylor, Fatemeh Geranmayeh
4228                                    </span>
4229                                </p>
4230                            </a>
4231                            <a class="w3-text" href="wang24e_interspeech.html">
4232                                <p>
4233                                    Automatic Prediction of Amyotrophic Lateral Sclerosis Progression using Longitudinal Speech Transformer
4234                                    <br>
4235                                    <span class="w3-text w3-text-theme">
4236                                        Liming Wang, Yuan Gong, Nauman Dawalatabad, Marco Vilela, Katerina Placek, Brian Tracey, Yishu Gong, Alan Premasiri, Fernando Vieira, James Glass
4237                                    </span>
4238                                </p>
4239                            </a>
4240                            <a class="w3-text" href="kothare24_interspeech.html">
4241                                <p>
4242                                    How Consistent are Speech-Based Biomarkers in Remote Tracking of ALS Disease Progression Across Languages? A Case Study of English and Dutch
4243                                    <br>
4244                                    <span class="w3-text w3-text-theme">
4245                                        Hardik Kothare, Michael Neumann, Cathy Zhang, Jackson Liscombe, Jordi W J van Unnik, Lianne C M Botman, Leonard H van den Berg, Ruben P A van Eijk, Vikram Ramanarayanan
4246                                    </span>
4247                                </p>
4248                            </a>
4249                            <a class="w3-text" href="spiesberger24_interspeech.html">
4250                                <p>
4251                                    “So . . . my child . . . ” – How Child ADHD Influences the Way Parents Talk
4252                                    <br>
4253                                    <span class="w3-text w3-text-theme">
4254                                        Anika A. Spiesberger, Andreas Triantafyllopoulos, Alexander Kathan, Anastasia Semertzidou, Caterina Gawrilow, Tilman Reinelt, Wolfgang A. Rauch, Björn Schuller
4255                                    </span>
4256                                </p>
4257                            </a>
4258                            <a class="w3-text" href="dineley24_interspeech.html">
4259                                <p>
4260                                    Variability of speech timing features across repeated recordings: a comparison of open-source extraction techniques
4261                                    <br>
4262                                    <span class="w3-text w3-text-theme">
4263                                        Judith Dineley, Ewan Carr, Lauren L. White, Catriona Lucas, Zahia Rahman, Tian Pan, Faith Matcham, Johnny Downs, Richard J. Dobson, Thomas F. Quatieri, Nicholas Cummins
4264                                    </span>
4265                                </p>
4266                            </a>
4267                            <a class="w3-text" href="labrak24_interspeech.html">
4268                                <p>
4269                                    Zero-Shot End-To-End Spoken Question Answering In Medical Domain
4270                                    <br>
4271                                    <span class="w3-text w3-text-theme">
4272                                        Yanis Labrak, Adel Moumen, Richard Dufour, Mickael Rouvier
4273                                    </span>
4274                                </p>
4275                            </a>
4276                            <a class="w3-text" href="jiang24b_interspeech.html">
4277                                <p>
4278                                    Perceiver-Prompt: Flexible Speaker Adaptation in Whisper for Chinese Disordered Speech Recognition
4279                                    <br>
4280                                    <span class="w3-text w3-text-theme">
4281                                        Yicong Jiang, Tianzi Wang, Xurong Xie, Juan Liu, Wei Sun, Nan Yan, Hui Chen, Lan Wang, Xunying Liu, Feng Tian
4282                                    </span>
4283                                </p>
4284                            </a>
4285                        </div>
4286                    </div>
4287                    <br>
4288                    <div class="w3-content" style="height:10px"  id="Show and Tell 2"></div>
4289                    <div class="w3-card w3-round w3-white w3-padding">
4290                        <div class="w3-container"  style="margin-top:40px">
4291                            <h4 class="w3-center">Show and Tell 2</h4>
4292                            <hr>
4293                            <a class="w3-text" href="v24_interspeech.html">
4294                                <p>
4295                                    Custom wake word detection
4296                                    <br>
4297                                    <span class="w3-text w3-text-theme">
4298                                        Kesavaraj V, Charan Devarkonda, Vamshiraghusimha Narasinga, Anil Kumar Vuppala
4299                                    </span>
4300                                </p>
4301                            </a>
4302                            <a class="w3-text" href="chen24z_interspeech.html">
4303                                <p>
4304                                    Edged based audio-visual speech enhancement demonstrator
4305                                    <br>
4306                                    <span class="w3-text w3-text-theme">
4307                                        Song Chen, Mandar Gogate, Kia Dashtipour, Jasper Kirton-Wingate, Adeel Hussain, Faiyaz Doctor, Tughrul Arslan, Amir Hussain
4308                                    </span>
4309                                </p>
4310                            </a>
4311                            <a class="w3-text" href="anway24_interspeech.html">
4312                                <p>
4313                                    Real-Time Gaze-directed speech enhancement for audio-visual hearing-aids
4314                                    <br>
4315                                    <span class="w3-text w3-text-theme">
4316                                        Arif Reza Anway, Bryony Buck, Mandar Gogate, Kia Dashtipour, Michael Akeroyd, Amir Hussain
4317                                    </span>
4318                                </p>
4319                            </a>
4320                            <a class="w3-text" href="kumar24c_interspeech.html">
4321                                <p>
4322                                    Detection of background agents speech in contact centers
4323                                    <br>
4324                                    <span class="w3-text w3-text-theme">
4325                                        Abhishek Kumar, Srikanth Konjeti, Jithendra Vepa
4326                                    </span>
4327                                </p>
4328                            </a>
4329                            <a class="w3-text" href="koilakuntla24_interspeech.html">
4330                                <p>
4331                                    Leveraging large language models for post-transcription correction in contact centers
4332                                    <br>
4333                                    <span class="w3-text w3-text-theme">
4334                                        Bramhendra Koilakuntla, Prajesh Rana, Paras Ahuja, Srikanth Konjeti, Jithendra Vepa
4335                                    </span>
4336                                </p>
4337                            </a>
4338                            <a class="w3-text" href="schade24_interspeech.html">
4339                                <p>
4340                                    Understanding “understanding”: presenting a richly annotated multimodal corpus of dyadic interaction
4341                                    <br>
4342                                    <span class="w3-text w3-text-theme">
4343                                        Leonie Schade, Nico Dallmann, Olcay Tük, Stefan Lazarov, Petra Wagner
4344                                    </span>
4345                                </p>
4346                            </a>
4347                            <a class="w3-text" href="possamaidemenezes24_interspeech.html">
4348                                <p>
4349                                    A demonstrator for articulation-based command word recognition
4350                                    <br>
4351                                    <span class="w3-text w3-text-theme">
4352                                        Joao Vitor Possamai de Menezes, Arne-Lukas Fietkau, Tom Diener, Steffen Kurbis, Peter Birkholz
4353                                    </span>
4354                                </p>
4355                            </a>
4356                            <a class="w3-text" href="ward24b_interspeech.html">
4357                                <p>
4358                                    Pragmatically similar utterance finder demonstration
4359                                    <br>
4360                                    <span class="w3-text w3-text-theme">
4361                                        Nigel G. Ward, Andres Segura
4362                                    </span>
4363                                </p>
4364                            </a>
4365                            <a class="w3-text" href="liu24s_interspeech.html">
4366                                <p>
4367                                    Real-time scheme for rapid extraction of speaker embeddings in challenging recording conditions
4368                                    <br>
4369                                    <span class="w3-text w3-text-theme">
4370                                        Kai Liu, Ziqing Du, Zhou Huan, Xucheng Wan, Naijun Zheng
4371                                    </span>
4372                                </p>
4373                            </a>
4374                            <a class="w3-text" href="chen24aa_interspeech.html">
4375                                <p>
4376                                    TEEMI: a speaking practice tool for L2 English learners
4377                                    <br>
4378                                    <span class="w3-text w3-text-theme">
4379                                        Szu-Yu Chen, Tien-Hong Lo, Yao-Ting Sung, Ching-Yu Tseng, Berlin Chen
4380                                    </span>
4381                                </p>
4382                            </a>
4383                        </div>
4384                    </div>
4385                    <br>
4386                    <div class="w3-content" style="height:10px"  id="Prosody"></div>
4387                    <div class="w3-card w3-round w3-white w3-padding">
4388                        <div class="w3-container"  style="margin-top:40px">
4389                            <h4 class="w3-center">Prosody</h4>
4390                            <hr>
4391                            <a class="w3-text" href="hu24b_interspeech.html">
4392                                <p>
4393                                    Automatic pitch accent classification through image classification
4394                                    <br>
4395                                    <span class="w3-text w3-text-theme">
4396                                        Na Hu, Hugo Schnack, Amalia Arvaniti
4397                                    </span>
4398                                </p>
4399                            </a>
4400                            <a class="w3-text" href="geng24_interspeech.html">
4401                                <p>
4402                                    Form and Function in Prosodic Representation:  In the Case of 'ma' in Tianjin Mandarin
4403                                    <br>
4404                                    <span class="w3-text w3-text-theme">
4405                                        Tianqi Geng, Hui Feng
4406                                    </span>
4407                                </p>
4408                            </a>
4409                            <a class="w3-text" href="chakraborty24_interspeech.html">
4410                                <p>
4411                                    On Comparing Time- and Frequency-Domain Rhythm Measures in Classifying Assamese Dialects
4412                                    <br>
4413                                    <span class="w3-text w3-text-theme">
4414                                        Joyshree Chakraborty, Leena Dihingia, Priyankoo Sarmah, Rohit Sinha
4415                                    </span>
4416                                </p>
4417                            </a>
4418                            <a class="w3-text" href="riegger24_interspeech.html">
4419                                <p>
4420                                    The prosody of the verbal prefix ge-: historical and experimental evidence
4421                                    <br>
4422                                    <span class="w3-text w3-text-theme">
4423                                        Chiara Riegger, Tina Bögel, George Walkden
4424                                    </span>
4425                                </p>
4426                            </a>
4427                            <a class="w3-text" href="wu24n_interspeech.html">
4428                                <p>
4429                                    Influences of Morphosyntax and Semantics on the Intonation of Mandarin Chinese Wh-indeterminates
4430                                    <br>
4431                                    <span class="w3-text w3-text-theme">
4432                                        Hongchen Wu, Jiwon Yun
4433                                    </span>
4434                                </p>
4435                            </a>
4436                            <a class="w3-text" href="mumtaz24_interspeech.html">
4437                                <p>
4438                                    Urdu Alternative Questions: A Hat Pattern
4439                                    <br>
4440                                    <span class="w3-text w3-text-theme">
4441                                        Benazir Mumtaz, Miriam Butt
4442                                    </span>
4443                                </p>
4444                            </a>
4445                        </div>
4446                    </div>
4447                    <br>
4448                    <div class="w3-content" style="height:10px"  id="Foundational Models for Deepfake and Spoofed Speech Detection"></div>
4449                    <div class="w3-card w3-round w3-white w3-padding">
4450                        <div class="w3-container"  style="margin-top:40px">
4451                            <h4 class="w3-center">Foundational Models for Deepfake and Spoofed Speech Detection</h4>
4452                            <hr>
4453                            <a class="w3-text" href="tran24_interspeech.html">
4454                                <p>
4455                                    Spoofed Speech Detection with a Focus on Speaker Embedding
4456                                    <br>
4457                                    <span class="w3-text w3-text-theme">
4458                                        Hoan My Tran, David Guennec, Philippe Martin, Aghilas Sini, Damien Lolive, Arnaud Delhay, Pierre-François Marteau
4459                                    </span>
4460                                </p>
4461                            </a>
4462                            <a class="w3-text" href="martindonas24_interspeech.html">
4463                                <p>
4464                                    Exploring Self-supervised Embeddings and Synthetic Data Augmentation for Robust Audio Deepfake Detection
4465                                    <br>
4466                                    <span class="w3-text w3-text-theme">
4467                                        Juan M. Martín-Doñas, Aitor Álvarez, Eros Rosello, Angel M. Gomez, Antonio M. Peinado
4468                                    </span>
4469                                </p>
4470                            </a>
4471                            <a class="w3-text" href="pan24c_interspeech.html">
4472                                <p>
4473                                    Attentive Merging of Hidden Embeddings from Pre-trained Speech Model for Anti-spoofing Detection
4474                                    <br>
4475                                    <span class="w3-text w3-text-theme">
4476                                        Zihan Pan, Tianchi Liu, Hardik B. Sailor, Qiongqiong Wang
4477                                    </span>
4478                                </p>
4479                            </a>
4480                            <a class="w3-text" href="wu24c_interspeech.html">
4481                                <p>
4482                                    Adapter Learning from Pre-trained Model for Robust Spoof Speech Detection
4483                                    <br>
4484                                    <span class="w3-text w3-text-theme">
4485                                        Haochen Wu, Wu Guo, Shengyu Peng, Zhuhai Li, Jie Zhang
4486                                    </span>
4487                                </p>
4488                            </a>
4489                            <a class="w3-text" href="liu24b_interspeech.html">
4490                                <p>
4491                                    Speech Formants Integration for Generalized Detection of Synthetic Speech Spoofing Attacks
4492                                    <br>
4493                                    <span class="w3-text w3-text-theme">
4494                                        Kexu Liu, Yuanxin Wang, Shengchen Li, Xi Shao
4495                                    </span>
4496                                </p>
4497                            </a>
4498                            <a class="w3-text" href="doan24_interspeech.html">
4499                                <p>
4500                                    Balance, Multiple Augmentation, and Re-synthesis: A Triad Training Strategy for Enhanced Audio Deepfake Detection
4501                                    <br>
4502                                    <span class="w3-text w3-text-theme">
4503                                        Thien-Phuc Doan, Long Nguyen-Vu, Kihun Hong, Souhwan Jung
4504                                    </span>
4505                                </p>
4506                            </a>
4507                        </div>
4508                    </div>
4509                    <br>
4510                    <div class="w3-content" style="height:10px"  id="Speaker Recognition 1"></div>
4511                    <div class="w3-card w3-round w3-white w3-padding">
4512                        <div class="w3-container"  style="margin-top:40px">
4513                            <h4 class="w3-center">Speaker Recognition 1</h4>
4514                            <hr>
4515                            <a class="w3-text" href="peng24_interspeech.html">
4516                                <p>
4517                                    Fine-tune Pre-Trained Models with Multi-Level Feature Fusion for Speaker Verification
4518                                    <br>
4519                                    <span class="w3-text w3-text-theme">
4520                                        Shengyu Peng, Wu Guo, Haochen Wu, Zuoliang Li, Jie Zhang
4521                                    </span>
4522                                </p>
4523                            </a>
4524                            <a class="w3-text" href="yu24_interspeech.html">
4525                                <p>
4526                                    Speaker Conditional Sinc-Extractor for Personal VAD
4527                                    <br>
4528                                    <span class="w3-text w3-text-theme">
4529                                        En-Lun Yu, Kuan-Hsun Ho, Jeih-weih Hung, Shih-Chieh Huang, Berlin Chen
4530                                    </span>
4531                                </p>
4532                            </a>
4533                            <a class="w3-text" href="liou24_interspeech.html">
4534                                <p>
4535                                    Enhancing ECAPA-TDNN with Feature Processing Module and Attention Mechanism for Speaker Verification
4536                                    <br>
4537                                    <span class="w3-text w3-text-theme">
4538                                        Shiu-Hsiang Liou, Po-Cheng Chan, Chia-Ping Chen, Tzu-Chieh Lin, Chung-Li Lu, Yu-Han Cheng, Hsiang-Feng Chuang, Wei-Yu Chen
4539                                    </span>
4540                                </p>
4541                            </a>
4542                            <a class="w3-text" href="kim24j_interspeech.html">
4543                                <p>
4544                                    MR-RawNet: Speaker verification system with multiple temporal resolutions for variable duration utterances using raw waveforms
4545                                    <br>
4546                                    <span class="w3-text w3-text-theme">
4547                                        Seung-bin Kim, Chan-yeong Lim, Jungwoo Heo, Ju-ho Kim, Hyun-seo Shin, Kyo-Won K
4547oo, Ha-Jin Yu
4548                                    </span>
4549                                </p>
4550                            </a>
4551                            <a class="w3-text" href="nam24b_interspeech.html">
4552                                <p>
4553                                    Disentangled Representation Learning for Environment-agnostic Speaker Recognition
4554                                    <br>
4555                                    <span class="w3-text w3-text-theme">
4556                                        KiHyun Nam, Hee-Soo Heo, Jee-weon Jung, Joonson Chung
4557                                    </span>
4558                                </p>
4559                            </a>
4560                            <a class="w3-text" href="mosner24_interspeech.html">
4561                                <p>
4562                                    Multi-Channel Extension of Pre-trained Models for Speaker Verification
4563                                    <br>
4564                                    <span class="w3-text w3-text-theme">
4565                                        Ladislav Mošner, Romain Serizel, Lukáš Burget, Oldřich Plchot, Emmanuel Vincent, Junyi Peng, Jan Černocký
4566                                    </span>
4567                                </p>
4568                            </a>
4569                            <a class="w3-text" href="li24la_interspeech.html">
4570                                <p>
4571                                    Efficient Integrated Features Based on Pre-trained Models for Speaker Verification
4572                                    <br>
4573                                    <span class="w3-text w3-text-theme">
4574                                        Yishuang Li, Wenhao Guan, Hukai Huang, Shiyu Miao, Qi Su, Lin Li, Qingyang Hong
4575                                    </span>
4576                                </p>
4577                            </a>
4578                            <a class="w3-text" href="wang24ma_interspeech.html">
4579                                <p>
4580                                    SE/BN Adapter: Parametric Efficient Domain Adaptation for Speaker Recognition
4581                                    <br>
4582                                    <span class="w3-text w3-text-theme">
4583                                        Tianhao Wang, Lantian Li, Dong Wang
4584                                    </span>
4585                                </p>
4586                            </a>
4587                            <a class="w3-text" href="xie24b_interspeech.html">
4588                                <p>
4589                                    DB-PMAE: Dual-Branch Prototypical Masked AutoEncoder with locality for domain robust speaker verification
4590                                    <br>
4591                                    <span class="w3-text w3-text-theme">
4592                                        Wei-lin Xie, Yu-Xuan Xi, Yan Song, Jian-tao Zhang, Hao-yu Song, Ian McLoughlin
4593                                    </span>
4594                                </p>
4595                            </a>
4596                            <a class="w3-text" href="maciejewski24_interspeech.html">
4597                                <p>
4598                                    Evaluating the Santa Barbara Corpus: Challenges of the Breadth of Conversational Spoken Language
4599                                    <br>
4600                                    <span class="w3-text w3-text-theme">
4601                                        Matthew Maciejewski, Dominik Klement, Ruizhe Huang, Matthew Wiesner, Sanjeev Khudanpur
4602                                    </span>
4603                                </p>
4604                            </a>
4605                            <a class="w3-text" href="zhou24f_interspeech.html">
4606                                <p>
4607                                    A Comprehensive Investigation on Speaker Augmentation for Speaker Recognition
4608                                    <br>
4609                                    <span class="w3-text w3-text-theme">
4610                                        Zhenyu Zhou, Shibiao Xu, Shi Yin, Lantian Li, Dong Wang
4611                                    </span>
4612                                </p>
4613                            </a>
4614                        </div>
4615                    </div>
4616                    <br>
4617                    <div class="w3-content" style="height:10px"  id="Source Separation 1"></div>
4618                    <div class="w3-card w3-round w3-white w3-padding">
4619                        <div class="w3-container"  style="margin-top:40px">
4620                            <h4 class="w3-center">Source Separation 1</h4>
4621                            <hr>
4622                            <a class="w3-text" href="wang24i_interspeech.html">
4623                                <p>
4624                                    Noise-robust Speech Separation with Fast Generative Correction
4625                                    <br>
4626                                    <span class="w3-text w3-text-theme">
4627                                        Helin Wang, Jesús Villalba, Laureano Moro-Velazquez, Jiarui Hai, Thomas Thebaud, Najim Dehak
4628                                    </span>
4629                                </p>
4630                            </a>
4631                            <a class="w3-text" href="hartanto24_interspeech.html">
4632                                <p>
4633                                    MSDET: Multitask Speaker Separation and Direction-of-Arrival Estimation Training
4634                                    <br>
4635                                    <span class="w3-text w3-text-theme">
4636                                        Roland Hartanto, Sakriani Sakti, Koichi Shinoda
4637                                    </span>
4638                                </p>
4639                            </a>
4640                            <a class="w3-text" href="kealey24_interspeech.html">
4641                                <p>
4642                                    Unsupervised Improved MVDR Beamforming for Sound Enhancement
4643                                    <br>
4644                                    <span class="w3-text w3-text-theme">
4645                                        Jacob Kealey, John R. Hershey, François Grondin
4646                                    </span>
4647                                </p>
4648                            </a>
4649                            <a class="w3-text" href="chen24h_interspeech.html">
4650                                <p>
4651                                    Improving Generalization of Speech Separation in Real-World Scenarios: Strategies in Simulation, Optimization, and Evaluation
4652                                    <br>
4653                                    <span class="w3-text w3-text-theme">
4654                                        Ke Chen, Jiaqi Su, Taylor Berg-Kirkpatrick, Shlomo Dubnov, Zeyu Jin
4655                                    </span>
4656                                </p>
4657                            </a>
4658                            <a class="w3-text" href="kim24m_interspeech.html">
4659                                <p>
4660                                    Enhanced Deep Speech Separation in Clustered Ad Hoc Distributed Microphone Environments
4661                                    <br>
4662                                    <span class="w3-text w3-text-theme">
4663                                        Jihyun Kim, Stijn Kindt, Nilesh Madhu, Hong-Goo Kang
4664                                    </span>
4665                                </p>
4666                            </a>
4667                            <a class="w3-text" href="yip24_interspeech.html">
4668                                <p>
4669                                    Towards Audio Codec-based Speech Separation
4670                                    <br>
4671                                    <span class="w3-text w3-text-theme">
4672                                        Jia Qi Yip, Shengkui Zhao, Dianwen Ng, Eng Siong Chng, Bin Ma
4673                                    </span>
4674                                </p>
4675                            </a>
4676                        </div>
4677                    </div>
4678                    <br>
4679                    <div class="w3-content" style="height:10px"  id="Audio-Visual and Generative Speech Enhancement"></div>
4680                    <div class="w3-card w3-round w3-white w3-padding">
4681                        <div class="w3-container"  style="margin-top:40px">
4682                            <h4 class="w3-center">Audio-Visual and Generative Speech Enhancement</h4>
4683                            <hr>
4684                            <a class="w3-text" href="li24d_interspeech.html">
4685                                <p>
4686                                    Locally Aligned Rectified Flow Model for Speech Enhancement Towards Single-Step Diffusion
4687                                    <br>
4688                                    <span class="w3-text w3-text-theme">
4689                                        Zhengxiao Li, Nakamasa Inoue
4690                                    </span>
4691                                </p>
4692                            </a>
4693                            <a class="w3-text" href="wang24m_interspeech.html">
4694                                <p>
4695                                    Diffusion Gaussian Mixture Audio Denoise
4696                                    <br>
4697                                    <span class="w3-text w3-text-theme">
4698                                        Pu Wang, Junhui Li, Jialu Li, Liangdong Guo, Youshan Zhang
4699                                    </span>
4700                                </p>
4701                            </a>
4702                            <a class="w3-text" href="lay24_interspeech.html">
4703                                <p>
4704                                    An Analysis of the Variance of Diffusion-based Speech Enhancement
4705                                    <br>
4706                                    <span class="w3-text w3-text-theme">
4707                                        Bunlong Lay, Timo Gerkmann
4708                                    </span>
4709                                </p>
4710                            </a>
4711                            <a class="w3-text" href="jung24b_interspeech.html">
4712                                <p>
4713                                    FlowAVSE: Efficient Audio-Visual Speech Enhancement with Conditional Flow Matching
4714                                    <br>
4715                                    <span class="w3-text w3-text-theme">
4716                                        Chaeyoung Jung, Suyeon Lee, Ji-Hoon Kim, Joon Son Chung
4717                                    </span>
4718                                </p>
4719                            </a>
4720                            <a class="w3-text" href="chen24g_interspeech.html">
4721                                <p>
4722                                    RT-LA-VocE: Real-Time Low-SNR Audio-Visual Speech Enhancement
4723                                    <br>
4724                                    <span class="w3-text w3-text-theme">
4725                                        Honglie Chen, Rodrigo Mira, Stavros Petridis, Maja Pantic
4726                                    </span>
4727                                </p>
4728                            </a>
4729                            <a class="w3-text" href="li24m_interspeech.html">
4730                                <p>
4731                                    Complex Image-Generative Diffusion Transformer for Audio Denoising
4732                                    <br>
4733                                    <span class="w3-text w3-text-theme">
4734                                        Junhui Li, Pu Wang, Jialu Li, Youshan Zhang
4735                                    </span>
4736                                </p>
4737                            </a>
4738                            <a class="w3-text" href="hu24c_interspeech.html">
4739                                <p>
4740                                    Noise-aware Speech Enhancement using Diffusion Probabilistic Model
4741                                    <br>
4742                                    <span class="w3-text w3-text-theme">
4743                                        Yuchen Hu, Chen Chen, Ruizhe Li, Qiushi Zhu, Eng Siong Chng
4744                                    </span>
4745                                </p>
4746                            </a>
4747                        </div>
4748                    </div>
4749                    <br>
4750                    <div class="w3-content" style="height:10px"  id="Speech Privacy and Bandwidth Expansion"></div>
4751                    <div class="w3-card w3-round w3-white w3-padding">
4752                        <div class="w3-container"  style="margin-top:40px">
4753                            <h4 class="w3-center">Speech Privacy and Bandwidth Expansion</h4>
4754                            <hr>
4755                            <a class="w3-text" href="vali24_interspeech.html">
4756                                <p>
4757                                    Privacy PORCUPINE: Anonymization of Speaker Attributes Using Occurrence Normalization for Space-Filling Vector Quantization
4758                                    <br>
4759                                    <span class="w3-text w3-text-theme">
4760                                        Mohammad Hassan Vali, Tom Bäckström
4761                                    </span>
4762                                </p>
4763                            </a>
4764                            <a class="w3-text" href="singh24_interspeech.html">
4765                                <p>
4766                                    SilentCipher: Deep Audio Watermarking
4767                                    <br>
4768                                    <span class="w3-text w3-text-theme">
4769                                        Mayank Kumar Singh, Naoya Takahashi, Weihsiang Liao, Yuki Mitsufuji
4770                                    </span>
4771                                </p>
4772                            </a>
4773                            <a class="w3-text" href="fan24_interspeech.html">
4774                                <p>
4775                                    Frequency-mix Knowledge Distillation for Fake Speech Detection
4776                                    <br>
4777                                    <span class="w3-text w3-text-theme">
4778                                        Cunhang Fan, Shunbo Dong, Jun Xue, Yujie Chen, Jiangyan Yi, Zhao Lv
4779                                    </span>
4780                                </p>
4781                            </a>
4782                            <a class="w3-text" href="muller24_interspeech.html">
4783                                <p>
4784                                    A New Approach to Voice Authenticity
4785                                    <br>
4786                                    <span class="w3-text w3-text-theme">
4787                                        Nicolas M. Müller, Piotr Kawa, Shen Hu, Matthias Neu, Jennifer Williams, Philip Sperl, Konstantin Böttinger
4788                                    </span>
4789                                </p>
4790                            </a>
4791                            <a class="w3-text" href="zhou24b_interspeech.html">
4792                                <p>
4793                                    TraceableSpeech: Towards Proactively Traceable Text-to-Speech with Watermarking
4794                                    <br>
4795                                    <span class="w3-text w3-text-theme">
4796                                        Junzuo Zhou, Jiangyan Yi, Tao Wang, Jianhua Tao, Ye Bai, Chu Yuan Zhang, Yong Ren, Zhengqi Wen
4797                                    </span>
4798                                </p>
4799                            </a>
4800                            <a class="w3-text" href="liu24g_interspeech.html">
4801                                <p>
4802                                    HarmoNet: Partial DeepFake Detection Network based on Multi-scale HarmoF0 Feature Fusion
4803                                    <br>
4804                                    <span class="w3-text w3-text-theme">
4805                                        Liwei Liu, Huihui Wei, Dongya Liu, Zhonghua Fu
4806                                    </span>
4807                                </p>
4808                            </a>
4809                            <a class="w3-text" href="moussa24_interspeech.html">
4810                                <p>
4811                                    Unmasking Neural Codecs: Forensic Identification of AI-compressed Speech
4812                                    <br>
4813                                    <span class="w3-text w3-text-theme">
4814                                        Denise Moussa, Sandra Bergmann, Christian Riess
4815                                    </span>
4816                                </p>
4817                            </a>
4818                            <a class="w3-text" href="lin24c_interspeech.html">
4819                                <p>
4820                                    SWiBE: A Parameterized Stochastic Diffusion Process for Noise-Robust Bandwidth Expansion
4821                                    <br>
4822                                    <span class="w3-text w3-text-theme">
4823                                        Yin-Tse Lin, Shreya G. Upadhyay, Bo-Hao Su, Chi-Chun Lee
4824                                    </span>
4825                                </p>
4826                            </a>
4827                            <a class="w3-text" href="lu24_interspeech.html">
4828                                <p>
4829                                    MultiStage Speech Bandwidth Extension with Flexible Sampling Rate Control
4830                                    <br>
4831                                    <span class="w3-text w3-text-theme">
4832                                        Ye-Xin Lu, Yang Ai, Zheng-Yan Sheng, Zhen-Hua Ling
4833                                    </span>
4834                                </p>
4835                            </a>
4836                            <a class="w3-text" href="li24ea_interspeech.html">
4837                                <p>
4838                                    MaskSR: Masked Language Model for Full-band Speech Restoration
4839                                    <br>
4840                                    <span class="w3-text w3-text-theme">
4841                                        Xu Li, Qirui Wang, Xiaoyu Liu
4842                                    </span>
4843                                </p>
4844                            </a>
4845                        </div>
4846                    </div>
4847                    <br>
4848                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Prosody"></div>
4849                    <div class="w3-card w3-round w3-white w3-padding">
4850                        <div class="w3-container"  style="margin-top:40px">
4851                            <h4 class="w3-center">Speech Synthesis: Prosody</h4>
4852                            <hr>
4853                            <a class="w3-text" href="korotkova24_interspeech.html">
4854                                <p>
4855                                    Word-level Text Markup for Prosody Control in Speech Synthesis
4856                                    <br>
4857                                    <span class="w3-text w3-text-theme">
4858                                        Yuliya Korotkova, Ilya Kalinovskiy, Tatiana Vakhrusheva
4859                                    </span>
4860                                </p>
4861                            </a>
4862                            <a class="w3-text" href="mehta24b_interspeech.html">
4863                                <p>
4864                                    Should you use a probabilistic duration model in TTS? Probably! Especially for spontaneous speech
4865                                    <br>
4866                                    <span class="w3-text w3-text-theme">
4867                                        Shivam Mehta, Harm Lameris, Rajiv Punmiya, Jonas Beskow, Eva Szekely, Gustav Eje Henter
4868                                    </span>
4869                                </p>
4870                            </a>
4871                            <a class="w3-text" href="eskimez24_interspeech.html">
4872                                <p>
4873                                    Total-Duration-Aware Duration Modeling for Text-to-Speech Systems
4874                                    <br>
4875                                    <span class="w3-text w3-text-theme">
4876                                        Sefik Emre Eskimez, Xiaofei Wang, Manthan Thakker, Chung-Hsien Tsai, Canrun Li, Zhen Xiao, Hemin Yang, Zirun Zhu, Min Tang, Jinyu Li, Sheng Zhao, Naoyuki Kanda
4877                                    </span>
4878                                </p>
4879                            </a>
4880                            <a class="w3-text" href="maurya24_interspeech.html">
4881                                <p>
4882                                    A Human-in-the-Loop Approach to Improving Cross-Text Prosody Transfer
4883                                    <br>
4884                                    <span class="w3-text w3-text-theme">
4885                                        Himanshu Maurya, Atli Sigurgeirsson
4886                                    </span>
4887                                </p>
4888                            </a>
4889                            <a class="w3-text" href="jiang24d_interspeech.html">
4890                                <p>
4891                                    Towards Expressive Zero-Shot Speech Synthesis with Hierarchical Prosody Modeling
4892                                    <br>
4893                                    <span class="w3-text w3-text-theme">
4894                                        Yuepeng Jiang, Tao Li, Fengyu Yang, Lei Xie, Meng Meng, Yujun Wang
4895                                    </span>
4896                                </p>
4897                            </a>
4898                            <a class="w3-text" href="zhong24c_interspeech.html">
4899                                <p>
4900                                    Multi-Modal Automatic Prosody Annotation with Contrastive Pretraining of Speech-Silence and Word-Punctuation
4901                                    <br>
4902                                    <span class="w3-text w3-text-theme">
4903                                        Jinzuomu Zhong, Yang Li, Hui Huang, Korin Richmond, Jie Liu, Zhiba Su, Jing Guo, Benlai Tang, Fengjie Zhu
4904                                    </span>
4905                                </p>
4906                            </a>
4907                        </div>
4908                    </div>
4909                    <br>
4910                    <div class="w3-content" style="height:10px"  id="Accented Speech, Prosodic Features, Dialect, Emotion, Sound Classification"></div>
4911                    <div class="w3-card w3-round w3-white w3-padding">
4912                        <div class="w3-container"  style="margin-top:40px">
4913                            <h4 class="w3-center">Accented Speech, Prosodic Features, Dialect, Emotion, Sound Classification</h4>
4914                            <hr>
4915                            <a class="w3-text" href="prabhu24b_interspeech.html">
4916                                <p>
4917                                    Improving Self-supervised Pre-training using Accent-Specific Codebooks
4918                                    <br>
4919                                    <span class="w3-text w3-text-theme">
4920                                        Darshan Prabhu, Abhishek Gupta, Omkar Nitsure, Preethi Jyothi, Sriram Ganapathy
4921                                    </span>
4922                                </p>
4923                            </a>
4924                            <a class="w3-text" href="afonja24_interspeech.html">
4925                                <p>
4926                                    Performant ASR Models for Medical Entities in Accented Speech
4927                                    <br>
4928                                    <span class="w3-text w3-text-theme">
4929                                        Tejumade Afonja, Tobi Olatunji, Sewade Ogun, Naome A. Etori, Abraham Owodunni, Moshood Yekini
4930                                    </span>
4931                                </p>
4932                            </a>
4933                            <a class="w3-text" href="javed24_interspeech.html">
4934                                <p>
4935                                    LAHAJA: A Robust Multi-accent Benchmark for Evaluating Hindi ASR Systems
4936                                    <br>
4937                                    <span class="w3-text w3-text-theme">
4938                                        Tahir Javed, Janki Nawale, Sakshi Joshi, Eldho George, Kaushal Bhogale, Deovrat Mehendale, Mitesh M. Khapra
4939                                    </span>
4940                                </p>
4941                            </a>
4942                            <a class="w3-text" href="kim24v_interspeech.html">
4943                                <p>
4944                                    LearnerVoice: A Dataset of Non-Native English Learners’ Spontaneous Speech
4945                                    <br>
4946                                    <span class="w3-text w3-text-theme">
4947                                        Haechan Kim, Junho Myung, Seoyoung Kim, Sungpah Lee, Dongyeop Kang, Juho Kim
4948                                    </span>
4949                                </p>
4950                            </a>
4951                            <a class="w3-text" href="lin24m_interspeech.html">
4952                                <p>
4953                                    MinSpeech: A Corpus of Southern Min Dialect for Automatic Speech Recognition
4954                                    <br>
4955                                    <span class="w3-text w3-text-theme">
4956                                        Jiayan Lin, Shenghui Lu, Hukai Huang, Wenhao Guan, Binbin Xu, Hui Bu, Qingyang Hong, Lin Li
4957                                    </span>
4958                                </p>
4959                            </a>
4960                            <a class="w3-text" href="hu24e_interspeech.html">
4961                                <p>
4962                                    Cross-modal Features Interaction-and-Aggregation Network with Self-consistency Training for Speech Emotion Recognition
4963                                    <br>
4964                                    <span class="w3-text w3-text-theme">
4965                                        Ying Hu, Huamin Yang, Hao Huang, Liang He
4966                                    </span>
4967                                </p>
4968                            </a>
4969                            <a class="w3-text" href="goel24_interspeech.html">
4970                                <p>
4971                                    Exploring Multilingual Unseen Speaker Emotion Recognition: Leveraging Co-Attention Cues in Multitask Learning
4972                                    <br>
4973                                    <span class="w3-text w3-text-theme">
4974                                        Arnav Goel, Medha Hira, Anubha Gupta
4975                                    </span>
4976                                </p>
4977                            </a>
4978                            <a class="w3-text" href="bukhari24_interspeech.html">
4979                                <p>
4980                                    SELM: Enhancing Speech Emotion Recognition for Out-of-Domain Scenarios
4981                                    <br>
4982                                    <span class="w3-text w3-text-theme">
4983                                        Hazim Bukhari, Soham Deshmukh, Hira Dhamyal, Bhiksha Raj, Rita Singh
4984                                    </span>
4985                                </p>
4986                            </a>
4987                            <a class="w3-text" href="bentum24_interspeech.html">
4988                                <p>
4989                                    The Processing of Stress in End-to-End Automatic Speech Recognition Models
4990                                    <br>
4991                                    <span class="w3-text w3-text-theme">
4992                                        Martijn Bentum, Louis ten Bosch, Tom Lentz
4993                                    </span>
4994                                </p>
4995                            </a>
4996                            <a class="w3-text" href="nguyen24b_interspeech.html">
4997                                <p>
4998                                    LingWav2Vec2: Linguistic-augmented wav2vec 2.0 for Vietnamese Mispronunciation Detection
4999                                    <br>
5000                                    <span class="w3-text w3-text-theme">
5001                                        Tuan Nguyen, Huy Dat Tran
5002                                    </span>
5003                                </p>
5004                            </a>
5005                            <a class="w3-text" href="mogridge24_interspeech.html">
5006                                <p>
5007                                    Learning from memory-based models
5008                                    <br>
5009                                    <span class="w3-text w3-text-theme">
5010                                        Rhiannon Mogridge, Anton Ragni
5011                                    </span>
5012                                </p>
5013                            </a>
5014                            <a class="w3-text" href="chen24s_interspeech.html">
5015                                <p>
5016                                    Towards End-to-End Unified Recognition for Mandarin and Cantonese
5017                                    <br>
5018                                    <span class="w3-text w3-text-theme">
5019                                        Meiling Chen, Pengjie Liu, Heng Yang, Haofeng Wang
5020                                    </span>
5021                                </p>
5022                            </a>
5023                        </div>
5024                    </div>
5025                    <br>
5026                    <div class="w3-content" style="height:10px"  id="Neural Network Adaptation"></div>
5027                    <div class="w3-card w3-round w3-white w3-padding">
5028                        <div class="w3-container"  style="margin-top:40px">
5029                            <h4 class="w3-center">Neural Network Adaptation</h4>
5030                            <hr>
5031                            <a class="w3-text" href="rolland24b_interspeech.html">
5032                                <p>
5033                                    Shared-Adapters: A Novel Transformer-based Parameter Efficient Transfer Learning Approach For Children’s Automatic Speech Recognition
5034                                    <br>
5035                                    <span class="w3-text w3-text-theme">
5036                                        Thomas Rolland, Alberto Abad
5037                                    </span>
5038                                </p>
5039                            </a>
5040                            <a class="w3-text" href="huo24_interspeech.html">
5041                                <p>
5042                                    AdaRA: Adaptive Rank Allocation of Residual Adapters for Speech Foundation Model
5043                                    <br>
5044                                    <span class="w3-text w3-text-theme">
5045                                        Zhouyuan Huo, Dongseong Hwang, Gan Song, Khe Chai Sim, Weiran Wang
5046                                    </span>
5047                                </p>
5048                            </a>
5049                            <a class="w3-text" href="shim24_interspeech.html">
5050                                <p>
5051                                    Leveraging Adapter for Parameter-Efficient ASR Encoder
5052                                    <br>
5053                                    <span class="w3-text w3-text-theme">
5054                                        Kyuhong Shim, Jinkyu Lee, Hyunjae Kim
5055                                    </span>
5056                                </p>
5057                            </a>
5058                            <a class="w3-text" href="kang24_interspeech.html">
5059                                <p>
5060                                    Whisper Multilingual Downstream Task Tuning Using Task Vectors
5061                                    <br>
5062                                    <span class="w3-text w3-text-theme">
5063                                        Ji-Hun Kang, Jae-Hong Lee, Mun-Hak Lee, Joon-Hyuk Chang
5064                                    </span>
5065                                </p>
5066                            </a>
5067                            <a class="w3-text" href="li24ka_interspeech.html">
5068                                <p>
5069                                    Speaker-Smoothed kNN Speaker Adaptation for End-to-End ASR
5070                                    <br>
5071                                    <span class="w3-text w3-text-theme">
5072                                        Shaojun Li, Daimeng Wei, Hengchao Shang, Jiaxin Guo, ZongYao Li, Zhanglin Wu, Zhiqiang Rao, Yuanchang Luo, Xianghui He, Hao Yang
5073                                    </span>
5074                                </p>
5075                            </a>
5076                            <a class="w3-text" href="chen24n_interspeech.html">
5077                                <p>
5078                                    Qifusion-Net: Layer-adapted Stream/Non-stream Model for End-to-End Multi-Accent Speech Recognition
5079                                    <br>
5080                                    <span class="w3-text w3-text-theme">
5081                                        Jinming Chen, Jingyi Fang, Yuanzhong Zheng, Yaoxuan Wang, Haojun Fei
5082                                    </span>
5083                                </p>
5084                            </a>
5085                        </div>
5086                    </div>
5087                    <br>
5088                    <div class="w3-content" style="height:10px"  id="ASR and LLMs"></div>
5089                    <div class="w3-card w3-round w3-white w3-padding">
5090                        <div class="w3-container"  style="margin-top:40px">
5091                            <h4 class="w3-center">ASR and LLMs</h4>
5092                            <hr>
5093                            <a class="w3-text" href="yoon24_interspeech.html">
5094                                <p>
5095                                    HuBERT-EE: Early Exiting HuBERT for Efficient Speech Recognition
5096                                    <br>
5097                                    <span class="w3-text w3-text-theme">
5098                                        Ji Won Yoon, Beom Jun Woo, Nam Soo Kim
5099                                    </span>
5100                                </p>
5101                            </a>
5102                            <a class="w3-text" href="yang24f_interspeech.html">
5103                                <p>
5104                                    MaLa-ASR: Multimedia-Assisted LLM-Based ASR
5105                                    <br>
5106                                    <span class="w3-text w3-text-theme">
5107                                        Guanrou Yang, Ziyang Ma, Fan Yu, Zhifu Gao, Shiliang Zhang, Xie Chen
5108                                    </span>
5109                                </p>
5110                            </a>
5111                            <a class="w3-text" href="choi24_interspeech.html">
5112                                <p>
5113                                    Spoken-to-written text conversion with Large Language Model
5114                                    <br>
5115                                    <span class="w3-text w3-text-theme">
5116                                        HyunJung Choi, Muyeol Choi, Yohan Lim, Minkyu Lee, Seonhui Kim, Seung Yun, Donghyun Kim, SangHun Kim
5117                                    </span>
5118                                </p>
5119                            </a>
5120                            <a class="w3-text" href="ai24_interspeech.html">
5121                                <p>
5122                                    MM-KWS: Multi-modal Prompts for Multilingual User-defined Keyword Spotting
5123                                    <br>
5124                                    <span class="w3-text w3-text-theme">
5125                                        Zhiqi Ai, Zhiyong Chen, Shugong Xu
5126                                    </span>
5127                                </p>
5128                            </a>
5129                            <a class="w3-text" href="rouditchenko24_interspeech.html">
5130                                <p>
5131                                    Whisper-Flamingo: Integrating Visual Features into Whisper for Audio-Visual Speech Recognition and Translation
5132                                    <br>
5133                                    <span class="w3-text w3-text-theme">
5134                                        Andrew Rouditchenko, Yuan Gong, Samuel Thomas, Leonid Karlinsky, Hilde Kuehne, Rogerio Feris, James Glass
5135                                    </span>
5136                                </p>
5137                            </a>
5138                            <a class="w3-text" href="prajwal24_interspeech.html">
5139                                <p>
5140                                    Speech Recognition Models are Strong Lip-readers
5141                                    <br>
5142                                    <span class="w3-text w3-text-theme">
5143                                        K R Prajwal, Triantafyllos Afouras, Andrew Zisserman
5144                                    </span>
5145                                </p>
5146                            </a>
5147                        </div>
5148                    </div>
5149                    <br>
5150                    <div class="w3-content" style="height:10px"  id="Pathological Speech Analysis 3"></div>
5151                    <div class="w3-card w3-round w3-white w3-padding">
5152                        <div class="w3-container"  style="margin-top:40px">
5153                            <h4 class="w3-center">Pathological Speech Analysis 3</h4>
5154                            <hr>
5155                            <a class="w3-text" href="baumann24b_interspeech.html">
5156                                <p>
5157                                    Towards Self-Attention Understanding for Automatic Articulatory Processes Analysis in Cleft Lip and Palate Speech
5158                                    <br>
5159                                    <span class="w3-text w3-text-theme">
5160                                        Ilja Baumann, Dominik Wagner, Maria Schuster, Korbinian Riedhammer, Elmar Noeth, Tobias Bocklet
5161                                    </span>
5162                                </p>
5163                            </a>
5164                            <a class="w3-text" href="liu24f_interspeech.html">
5165                                <p>
5166                                    Clever Hans Effect Found in Automatic Detection of Alzheimer's Disease through Speech
5167                                    <br>
5168                                    <span class="w3-text w3-text-theme">
5169                                        Yin-Long Liu, Rui Feng, Jia-Hong Yuan, Zhen-Hua Ling
5170                                    </span>
5171                                </p>
5172                            </a>
5173                            <a class="w3-text" href="lin24k_interspeech.html">
5174                                <p>
5175                                    Leveraging Phonemic Transcription and Whisper toward Clinically Significant Indices for Automatic Child Speech Assessment
5176                                    <br>
5177                                    <span class="w3-text w3-text-theme">
5178                                        Yeh-Sheng Lin, Shu-Chuan Tseng, Jyh-Shing Roger Jang
5179                                    </span>
5180                                </p>
5181                            </a>
5182                            <a class="w3-text" href="dang24b_interspeech.html">
5183                                <p>
5184                                    Developing vocal system impaired patient-aimed voice quality assessment approach using ASR representation-included multiple features
5185                                    <br>
5186                                    <span class="w3-text w3-text-theme">
5187                                        Shaoxiang Dang, Tetsuya Matsumoto, Yoshinori Takeuchi, Takashi Tsuboi, Yasuhiro Tanaka, Daisuke Nakatsubo, Satoshi Maesawa, Ryuta Saito, Masahisa Katsuno, Hiroaki Kudo
5188                                    </span>
5189                                </p>
5190                            </a>
5191                            <a class="w3-text" href="hsu24_interspeech.html">
5192                                <p>
5193                                    A Cluster-based Personalized Federated Learning Strategy for End-to-End ASR of Dementia Patients
5194                                    <br>
5195                                    <span class="w3-text w3-text-theme">
5196                                        Wei-Tung Hsu, Chin-Po Chen, Yun-Shao Lin, Chi-Chun Lee
5197                                    </span>
5198                                </p>
5199                            </a>
5200                            <a class="w3-text" href="kalabakov24_interspeech.html">
5201                                <p>
5202                                    A Comparative Analysis of Federated Learning for Speech-Based Cognitive Decline Detection
5203                                    <br>
5204                                    <span class="w3-text w3-text-theme">
5205                                        Stefan Kalabakov, Monica Gonzalez-Machorro, Florian Eyben, Björn W. Schuller, Bert Arnrich
5206                                    </span>
5207                                </p>
5208                            </a>
5209                            <a class="w3-text" href="neumann24_interspeech.html">
5210                                <p>
5211                                    Multimodal Digital Biomarkers for Longitudinal Tracking of Speech Impairment Severity in ALS: An Investigation of Clinically Important Differences
5212                                    <br>
5213                                    <span class="w3-text w3-text-theme">
5214                                        Michael Neumann, Hardik Kothare, Jackson Liscombe, Emma C.L. Leschly, Oliver Roesler, Vikram Ramanarayanan
5215                                    </span>
5216                                </p>
5217                            </a>
5218                        </div>
5219                    </div>
5220                    <br>
5221                    <div class="w3-content" style="height:10px"  id="Speech Disorders 3"></div>
5222                    <div class="w3-card w3-round w3-white w3-padding">
5223                        <div class="w3-container"  style="margin-top:40px">
5224                            <h4 class="w3-center">Speech Disorders 3</h4>
5225                            <hr>
5226                            <a class="w3-text" href="gao24c_interspeech.html">
5227                                <p>
5228                                    Enhancing Voice Wake-Up for Dysarthria: Mandarin Dysarthria Speech Corpus Release and Customized System Design
5229                                    <br>
5230                                    <span class="w3-text w3-text-theme">
5231                                        Ming Gao, Hang Chen, Jun Du, Xin Xu, Hongxiao Guo, Hui Bu, Jianxing Yang, Ming Li, Chin-Hui Lee
5232                                    </span>
5233                                </p>
5234                            </a>
5235                            <a class="w3-text" href="shah24_interspeech.html">
5236                                <p>
5237                                    Towards Improving NAM-to-Speech Synthesis Intelligibility using Self-Supervised Speech Models
5238                                    <br>
5239                                    <span class="w3-text w3-text-theme">
5240                                        Neil Shah, Shirish Karande, Vineet Gandhi
5241                                    </span>
5242                                </p>
5243                            </a>
5244                            <a class="w3-text" href="um24_interspeech.html">
5245                                <p>
5246                                    PARAN: Variational Autoencoder-based End-to-End Articulation-to-Speech System for Speech Intelligibility
5247                                    <br>
5248                                    <span class="w3-text w3-text-theme">
5249                                        Seyun Um, Doyeon Kim, Hong-Goo Kang
5250                                    </span>
5251                                </p>
5252                            </a>
5253                            <a class="w3-text" href="chen24b_interspeech.html">
5254                                <p>
5255                                    Acoustic changes in speech prosody produced by children with autism after robot-assisted speech training
5256                                    <br>
5257                                    <span class="w3-text w3-text-theme">
5258                                        Si Chen, Bruce Xiao Wang, Yitian Hong, Fang Zhou, Angel Chan, Po-yi Tang, Bin Li, Chunyi Wen, James Cheung, Yan Liu, Zhuoming Chen
5259                                    </span>
5260                                </p>
5261                            </a>
5262                            <a class="w3-text" href="zheng24c_interspeech.html">
5263                                <p>
5264                                    Fine-Tuning Automatic Speech Recognition for People with Parkinson's: An Effective Strategy for Enhancing Speech Technology Accessibility
5265                                    <br>
5266                                    <span class="w3-text w3-text-theme">
5267                                        Xiuwen Zheng, Bornali Phukon, Mark Hasegawa-Johnson
5268                                    </span>
5269                                </p>
5270                            </a>
5271                            <a class="w3-text" href="jiang24_interspeech.html">
5272                                <p>
5273                                    Learnings from curating a trustworthy, well-annotated, and useful dataset of disordered English speech
5274                                    <br>
5275                                    <span class="w3-text w3-text-theme">
5276                                        Pan-Pan Jiang, Jimmy Tobin, Katrin Tomanek, Robert MacDonald, Katie Seaver, Richard Cave, Marilyn Ladewig, Rus Heywood, Jordan Green
5277                                    </span>
5278                                </p>
5279                            </a>
5280                            <a class="w3-text" href="leung24_interspeech.html">
5281                                <p>
5282                                    Training Data Augmentation for Dysarthric Automatic Speech Recognition by Text-to-Dysarthric-Speech Synthesis
5283                                    <br>
5284                                    <span class="w3-text w3-text-theme">
5285                                        Wing-Zin Leung, Mattias Cross, Anton Ragni, Stefan Goetze
5286                                    </span>
5287                                </p>
5288                            </a>
5289                            <a class="w3-text" href="gosztolya24b_interspeech.html">
5290                                <p>
5291                                    Wav2vec 2.0 Embeddings Are No Swiss Army Knife -- A Case Study for Multiple Sclerosis
5292                                    <br>
5293                                    <span class="w3-text w3-text-theme">
5294                                        Gábor Gosztolya, Mercedes Vetráb, Veronika Svindt, Judit Bóna, Ildikó Hoffmann
5295                                    </span>
5296                                </p>
5297                            </a>
5298                        </div>
5299                    </div>
5300                    <br>
5301                    <div class="w3-content" style="height:10px"  id="Speech Recognition with Large Pretrained Speech Models for Under-represented Languages (Special Session)"></div>
5302                    <div class="w3-card w3-round w3-white w3-padding">
5303                        <div class="w3-container"  style="margin-top:40px">
5304                            <h4 class="w3-center">Speech Recognition with Large Pretrained Speech Models for Under-represented Languages (Special Session)</h4>
5305                            <hr>
5306                            <a class="w3-text" href="shih24_interspeech.html">
5307                                <p>
5308                                    Interface Design for Self-Supervised Speech Models
5309                                    <br>
5310                                    <span class="w3-text w3-text-theme">
5311                                        Yi-Jen Shih, David Harwath
5312                                    </span>
5313                                </p>
5314                            </a>
5315                            <a class="w3-text" href="xu24d_interspeech.html">
5316                                <p>
5317                                    Comparing Discrete and Continuous Space LLMs for Speech Recognition
5318                                    <br>
5319                                    <span class="w3-text w3-text-theme">
5320                                        Yaoxun Xu, Shi-Xiong Zhang, Jianwei Yu, Zhiyong Wu, Dong Yu
5321                                    </span>
5322                                </p>
5323                            </a>
5324                            <a class="w3-text" href="li24ia_interspeech.html">
5325                                <p>
5326                                    Improving Whisper's Recognition Performance for Under-Represented Language Kazakh Leveraging Unpaired Speech and Text
5327                                    <br>
5328                                    <span class="w3-text w3-text-theme">
5329                                        Jinpeng Li, Yu Pu, Qi Sun, Wei-Qiang Zhang
5330                                    </span>
5331                                </p>
5332                            </a>
5333                            <a class="w3-text" href="bhogale24_interspeech.html">
5334                                <p>
5335                                    Empowering Low-Resource Language ASR via Large-Scale Pseudo Labeling
5336                                    <br>
5337                                    <span class="w3-text w3-text-theme">
5338                                        Kaushal Santosh Bhogale, Deovrat Mehendale, Niharika Parasa, Sathish Kumar Reddy G, Tahir Javed, Pratyush Kumar, Mitesh M. Khapra
5339                                    </span>
5340                                </p>
5341                            </a>
5342                            <a class="w3-text" href="li24i_interspeech.html">
5343                                <p>
5344                                    Interleaved Audio/Audiovisual Transfer Learning for AV-ASR in Low-Resourced Languages
5345                                    <br>
5346                                    <span class="w3-text w3-text-theme">
5347                                        Zhengyang Li, Patrick Blumenberg, Jing Liu, Thomas Graave, Timo Lohrenz, Siegfried Kunzmann, Tim Fingscheidt
5348                                    </span>
5349                                </p>
5350                            </a>
5351                            <a class="w3-text" href="udupa24_interspeech.html">
5352                                <p>
5353                                    Adapter pre-training for improved speech recognition in unseen domains using low resource adapter tuning of self-supervised models
5354                                    <br>
5355                                    <span class="w3-text w3-text-theme">
5356                                        Sathvik Udupa, Jesuraj Bandekar, Saurabh Kumar, Deekshitha G, Sandhya B, Abhayjeet S, Savitha Murthy, Priyanka Pai, Srinivasa Raghavan, Raoul Nanavati, Prasanta Kumar Ghosh
5357                                    </span>
5358                                </p>
5359                            </a>
5360                            <a class="w3-text" href="xu24h_interspeech.html">
5361                                <p>
5362                                    Towards Rehearsal-Free Multilingual ASR: A LoRA-based Case Study on Whisper 
5363                                    <br>
5364                                    <span class="w3-text w3-text-theme">
5365                                        Tianyi Xu, Kaixun Huang, Pengcheng Guo, Yu Zhou, Longtao Huang, Hui Xue, Lei Xie
5366                                    </span>
5367                                </p>
5368                            </a>
5369                            <a class="w3-text" href="getman24b_interspeech.html">
5370                                <p>
5371                                    Exploring adaptation techniques of large speech foundation models for low-resource ASR: a case study on Northern Sámi
5372                                    <br>
5373                                    <span class="w3-text w3-text-theme">
5374                                        Yaroslav Getman, Tamas Grosz, Katri Hiovain-Asikainen, Mikko Kurimo
5375                                    </span>
5376                                </p>
5377                            </a>
5378                            <a class="w3-text" href="qian24_interspeech.html">
5379                                <p>
5380                                    Learn and Don't Forget: Adding a New Language to ASR Foundation Models
5381                                    <br>
5382                                    <span class="w3-text w3-text-theme">
5383                                        Mengjie Qian, Siyuan Tang, Rao Ma, Kate M. Knill, Mark J.F. Gales
5384                                    </span>
5385                                </p>
5386                            </a>
5387                        </div>
5388                    </div>
5389                    <br>
5390                    <div class="w3-content" style="height:10px"  id="Speech Processing Using Discrete Speech Units (Special Session)"></div>
5391                    <div class="w3-card w3-round w3-white w3-padding">
5392                        <div class="w3-container"  style="margin-top:40px">
5393                            <h4 class="w3-center">Speech Processing Using Discrete Speech Units (Special Session)</h4>
5394                            <hr>
5395                            <a class="w3-text" href="wu24q_interspeech.html">
5396                                <p>
5397                                    TokSing: Singing Voice Synthesis based on Discrete Tokens
5398                                    <br>
5399                                    <span class="w3-text w3-text-theme">
5400                                        Yuning Wu, Chunlei Zhang, Jiatong Shi, Yuxun Tang, Shan Yang, Qin Jin
5401                                    </span>
5402                                </p>
5403                            </a>
5404                            <a class="w3-text" href="mousavi24_interspeech.html">
5405                                <p>
5406                                    How Should We Extract Discrete Audio Tokens from Self-Supervised Models?
5407                                    <br>
5408                                    <span class="w3-text w3-text-theme">
5409                                        Pooneh Mousavi, Jarod Duret, Salah Zaiem, Luca Della Libera, Artem Ploujnikov, Cem Subakan, Mirco Ravanelli
5410                                    </span>
5411                                </p>
5412                            </a>
5413                            <a class="w3-text" href="chang24b_interspeech.html">
5414                                <p>
5415                                    The Interspeech 2024 Challenge on Speech Processing Using Discrete Units
5416                                    <br>
5417                                    <span class="w3-text w3-text-theme">
5418                                        Xuankai Chang, Jiatong Shi, Jinchuan Tian, Yuning Wu, Yuxun Tang, Yihan Wu, Shinji Watanabe, Yossi Adi, Xie Chen, Qin Jin
5419                                    </span>
5420                                </p>
5421                            </a>
5422                            <a class="w3-text" href="tang24c_interspeech.html">
5423                                <p>
5424                                    SingOMD: Singing Oriented Multi-resolution Discrete Representation Construction from Speech Models
5425                                    <br>
5426                                    <span class="w3-text w3-text-theme">
5427                                        Yuxun Tang, Yuning Wu, Jiatong Shi, Qin Jin
5428                                    </span>
5429                                </p>
5430                            </a>
5431                            <a class="w3-text" href="shi24h_interspeech.html">
5432                                <p>
5433                                    MMM: Multi-Layer Multi-Residual Multi-Stream Discrete Speech Representation from Self-supervised Learning Model
5434                                    <br>
5435                                    <span class="w3-text w3-text-theme">
5436                                        Jiatong Shi, Xutai Ma, Hirofumi Inaguma, Anna Sun, Shinji Watanabe
5437                                    </span>
5438                                </p>
5439                            </a>
5440                            <a class="w3-text" href="dhawan24_interspeech.html">
5441                                <p>
5442                                    Codec-ASR: Training Performant Automatic Speech Recognition Systems with Discrete Speech Representations
5443                                    <br>
5444                                    <span class="w3-text w3-text-theme">
5445                                        Kunal Dhawan, Nithin Rao Koluguri, Ante Jukić, Ryan Langman, Jagadeesh Balam, Boris Ginsburg
5446                                    </span>
5447                                </p>
5448                            </a>
5449                        </div>
5450                    </div>
5451                    <br>
5452                    <div class="w3-content" style="height:10px"  id="Keynote 3"></div>
5453                    <div class="w3-card w3-round w3-white w3-padding">
5454                        <div class="w3-container"  style="margin-top:40px">
5455                            <h4 class="w3-center">Keynote 3</h4>
5456                            <hr>
5457                            <a class="w3-text" href="noeth24_interspeech.html">
5458                                <p>
5459                                    Analysis of Pathological Speech – Pitfalls along the Way
5460                                    <br>
5461                                    <span class="w3-text w3-text-theme">
5462                                        Elmar Noeth
5463                                    </span>
5464                                </p>
5465                            </a>
5466                        </div>
5467                    </div>
5468                    <br>
5469                    <div class="w3-content" style="height:10px"  id="Databases and Progress in Methodology"></div>
5470                    <div class="w3-card w3-round w3-white w3-padding">
5471                        <div class="w3-container"  style="margin-top:40px">
5472                            <h4 class="w3-center">Databases and Progress in Methodology</h4>
5473                            <hr>
5474                            <a class="w3-text" href="ahn24b_interspeech.html">
5475                                <p>
5476                                    VoxSim: A perceptual voice similarity dataset
5477                                    <br>
5478                                    <span class="w3-text w3-text-theme">
5479                                        Junseok Ahn, Youkyum Kim, Yeunju Choi, Doyeop Kwak, Ji-Hoon Kim, Seongkyu Mun, Joon Son Chung
5480                                    </span>
5481                                </p>
5482                            </a>
5483                            <a class="w3-text" href="nijat24_interspeech.html">
5484                                <p>
5485                                    UY/CH-CHILD -- A Public Chinese L2 Speech Database of Uyghur Children
5486                                    <br>
5487                                    <span class="w3-text w3-text-theme">
5488                                        Mewlude Nijat, Chen Chen, Dong Wang, Askar Hamdulla
5489                                    </span>
5490                                </p>
5491                            </a>
5492                            <a class="w3-text" href="kumar24b_interspeech.html">
5493                                <p>
5494                                    State-of-the-art speech production MRI protocol for new 0.55 Tesla scanners
5495                                    <br>
5496                                    <span class="w3-text w3-text-theme">
5497                                        Prakash Kumar, Ye Tian, Yongwan Lim, Sophia X. Cui, Christina Hagedorn, Dani Byrd, Uttam K. Sinha, Shrikanth Narayanan, Krishna S. Nayak
5498                                    </span>
5499                                </p>
5500                            </a>
5501                            <a class="w3-text" href="shi24c_interspeech.html">
5502                                <p>
5503                                    DBD-CI: Doubling the Band Density for Bilateral Cochlear Implants
5504                                    <br>
5505                                    <span class="w3-text w3-text-theme">
5506                                        Mingyue Shi, Huali Zhou, Qinglin Meng, Nengheng Zheng
5507                                    </span>
5508                                </p>
5509                            </a>
5510                            <a class="w3-text" href="zhong24b_interspeech.html">
5511                                <p>
5512                                    Leveraging Large Language Models to Refine Automatic Feedback Generation at Articulatory Level in Computer Aided Pronunciation Training
5513                                    <br>
5514                                    <span class="w3-text w3-text-theme">
5515                                        Huihang Zhong, Yanlu Xie, ZiJin Yao
5516                                    </span>
5517                                </p>
5518                            </a>
5519                            <a class="w3-text" href="zhao24e_interspeech.html">
5520                                <p>
5521                                    Decoding Human Language Acquisition: EEG Evidence for Predictive Probabilistic Statistics in Word Segmentation
5522                                    <br>
5523                                    <span class="w3-text w3-text-theme">
5524                                        Bin Zhao, Mingxuan Huang, Chenlu Ma, Jinyi Xue, Aijun Li, Kunyu Xu
5525                                    </span>
5526                                </p>
5527                            </a>
5528                        </div>
5529                    </div>
5530                    <br>
5531                    <div class="w3-content" style="height:10px"  id="Articulation, Convergence and Perception"></div>
5532                    <div class="w3-card w3-round w3-white w3-padding">
5533                        <div class="w3-container"  style="margin-top:40px">
5534                            <h4 class="w3-center">Articulation, Convergence and Perception</h4>
5535                            <hr>
5536                            <a class="w3-text" href="giroud24_interspeech.html">
5537                                <p>
5538                                    Behavioral evidence for higher speech rate convergence following natural than artificial time altered speech
5539                                    <br>
5540                                    <span class="w3-text w3-text-theme">
5541                                        Jérémy Giroud, Jessica Lei, Kirsty Phillips, Matthew H. Davis
5542                                    </span>
5543                                </p>
5544                            </a>
5545                            <a class="w3-text" href="shen24c_interspeech.html">
5546                                <p>
5547                                    A novel experimental design for the study of listener-to-listener convergence in phoneme categorization
5548                                    <br>
5549                                    <span class="w3-text w3-text-theme">
5550                                        Qingye Shen, Leonardo Lancia, Noel Nguyen
5551                                    </span>
5552                                </p>
5553                            </a>
5554                            <a class="w3-text" href="li24ga_interspeech.html">
5555                                <p>
5556                                    Cross-Attention-Guided WaveNet for EEG-to-MEL Spectrogram Reconstruction
5557                                    <br>
5558                                    <span class="w3-text w3-text-theme">
5559                                        Hao Li, Yuan Fang, Xueliang Zhang, Fei Chen, Guanglai Gao
5560                                    </span>
5561                                </p>
5562                            </a>
5563                            <a class="w3-text" href="loddo24_interspeech.html">
5564                                <p>
5565                                    What if HAL breathed? Enhancing Empathy in Human-AI Interactions with Breathing Speech Synthesis
5566                                    <br>
5567                                    <span class="w3-text w3-text-theme">
5568                                        Nicolò Loddo, Francisca Pessanha, Almila Akdag
5569                                    </span>
5570                                </p>
5571                            </a>
5572                            <a class="w3-text" href="svenssonlundmark24_interspeech.html">
5573                                <p>
5574                                    Magnitude and timing of acceleration peaks in stressed and unstressed syllables
5575                                    <br>
5576                                    <span class="w3-text w3-text-theme">
5577                                        Malin Svensson Lundmark
5578                                    </span>
5579                                </p>
5580                            </a>
5581                        </div>
5582                    </div>
5583                    <br>
5584                    <div class="w3-content" style="height:10px"  id="Speech Emotion Recognition"></div>
5585                    <div class="w3-card w3-round w3-white w3-padding">
5586                        <div class="w3-container"  style="margin-top:40px">
5587                            <h4 class="w3-center">Speech Emotion Recognition</h4>
5588                            <hr>
5589                            <a class="w3-text" href="amiriparian24_interspeech.html">
5590                                <p>
5591                                    ExHuBERT: Enhancing HuBERT Through Block Extension and Fine-Tuning on 37 Emotion Datasets
5592                                    <br>
5593                                    <span class="w3-text w3-text-theme">
5594                                        Shahin Amiriparian, Filip Packań, Maurice Gerczuk, Björn W. Schuller
5595                                    </span>
5596                                </p>
5597                            </a>
5598                            <a class="w3-text" href="rittergutierrez24_interspeech.html">
5599                                <p>
5600                                    Dataset-Distillation Generative Model for Speech Emotion Recognition
5601                                    <br>
5602                                    <span class="w3-text w3-text-theme">
5603                                        Fabian Ritter-Gutierrez, Kuan-Po Huang, Jeremy H. M. Wong, Dianwen Ng, Hung-yi Lee, Nancy F. Chen, Eng-Siong Chng
5604                                    </span>
5605                                </p>
5606                            </a>
5607                            <a class="w3-text" href="mai24_interspeech.html">
5608                                <p>
5609                                    DropFormer: A Dynamic Noise-Dropping Transformer for Speech Emotion Recognition
5610                                    <br>
5611                                    <span class="w3-text w3-text-theme">
5612                                        Jialong Mai, Xiaofen Xing, Weidong Chen, Xiangmin Xu
5613                                    </span>
5614                                </p>
5615                            </a>
5616                            <a class="w3-text" href="niu24d_interspeech.html">
5617                                <p>
5618                                    From Text to Emotion: Unveiling the Emotion Annotation Capabilities of LLMs
5619                                    <br>
5620                                    <span class="w3-text w3-text-theme">
5621                                        Minxue Niu, Mimansa Jaiswal, Emily Mower Provost
5622                                    </span>
5623                                </p>
5624                            </a>
5625                        </div>
5626                    </div>
5627                    <br>
5628                    <div class="w3-content" style="height:10px"  id="Self-Supervised Models in Speaker Recognition"></div>
5629                    <div class="w3-card w3-round w3-white w3-padding">
5630                        <div class="w3-container"  style="margin-top:40px">
5631                            <h4 class="w3-center">Self-Supervised Models in Speaker Recognition</h4>
5632                            <hr>
5633                            <a class="w3-text" href="kim24c_interspeech.html">
5634                                <p>
5635                                    Self-supervised speaker verification with relational mask prediction
5636                                    <br>
5637                                    <span class="w3-text w3-text-theme">
5638                                        Ju-ho Kim, Hee-Soo Heo, Bong-Jin Lee, Youngki Kwon, Minjae Lee, Ha-Jin Yu
5639                                    </span>
5640                                </p>
5641                            </a>
5642                            <a class="w3-text" href="miara24_interspeech.html">
5643                                <p>
5644                                    Towards Supervised Performance on Speaker Verification with Self-Supervised Learning by Leveraging Large-Scale ASR Models
5645                                    <br>
5646                                    <span class="w3-text w3-text-theme">
5647                                        Victor Miara, Theo Lepage, Reda Dehak
5648                                    </span>
5649                                </p>
5650                            </a>
5651                            <a class="w3-text" href="lim24_interspeech.html">
5652                                <p>
5653                                    Improving Noise Robustness in Self-supervised Pre-trained Model for Speaker Verification
5654                                    <br>
5655                                    <span class="w3-text w3-text-theme">
5656                                        Chan-yeong Lim, Hyun-seo Shin, Ju-ho Kim, Jungwoo Heo, Kyo-Won Koo, Seung-bin Kim, Ha-Jin Yu
5657                                    </span>
5658                                </p>
5659                            </a>
5660                            <a class="w3-text" href="fathan24_interspeech.html">
5661                                <p>
5662                                    On the impact of several regularization techniques on label noise robustness of self-supervised speaker verification systems
5663                                    <br>
5664                                    <span class="w3-text w3-text-theme">
5665                                        Abderrahim Fathan, Xiaolin Zhu, Jahangir Alam
5666                                    </span>
5667                                </p>
5668                            </a>
5669                            <a class="w3-text" href="li24e_interspeech.html">
5670                                <p>
5671                                    Parameter-efficient Fine-tuning of Speaker-Aware Dynamic Prompts for Speaker Verification
5672                                    <br>
5673                                    <span class="w3-text w3-text-theme">
5674                                        Zhe Li, Man-wai Mak, Hung-yi Lee, Helen Meng
5675                                    </span>
5676                                </p>
5677                            </a>
5678                            <a class="w3-text" href="zhao24f_interspeech.html">
5679                                <p>
5680                                    Whisper-PMFA: Partial Multi-Scale Feature Aggregation for Speaker Verification using Whisper Models
5681                                    <br>
5682                                    <span class="w3-text w3-text-theme">
5683                                        Yiyang Zhao, Shuai Wang, Guangzhi Sun, Zehua Chen, Chao Zhang, Mingxing Xu, Thomas Fang Zheng
5684                                    </span>
5685                                </p>
5686                            </a>
5687                        </div>
5688                    </div>
5689                    <br>
5690                    <div class="w3-content" style="height:10px"  id="Speech Quality Assessment"></div>
5691                    <div class="w3-card w3-round w3-white w3-padding">
5692                        <div class="w3-container"  style="margin-top:40px">
5693                            <h4 class="w3-center">Speech Quality Assessment</h4>
5694                            <hr>
5695                            <a class="w3-text" href="hu24d_interspeech.html">
5696                                <p>
5697                                    Embedding Learning for Preference-based Speech Quality Assessment
5698                                    <br>
5699                                    <span class="w3-text w3-text-theme">
5700                                        ChengHung Hu, Yusuke Yasuda, Tomoki Toda
5701                                    </span>
5702                                </p>
5703                            </a>
5704                            <a class="w3-text" href="udupa24b_interspeech.html">
5705                                <p>
5706                                    IndicMOS: Multilingual MOS Prediction for 7 Indian languages
5707                                    <br>
5708                                    <span class="w3-text w3-text-theme">
5709                                        Sathvik Udupa, Soumi Maiti, Prasanta Kumar Ghosh
5710                                    </span>
5711                                </p>
5712                            </a>
5713                            <a class="w3-text" href="wells24_interspeech.html">
5714                                <p>
5715                                    Experimental evaluation of MOS, AB and BWS listening test designs
5716                                    <br>
5717                                    <span class="w3-text w3-text-theme">
5718                                        Dan Wells, Andrea Lorena Aldana Blanco, Cassia Valentini, Erica Cooper, Aidan Pine, Junichi Yamagishi, Korin Richmond
5719                                    </span>
5720                                </p>
5721                            </a>
5722                            <a class="w3-text" href="ta24_interspeech.html">
5723                                <p>
5724                                    Enhancing No-Reference Speech Quality Assessment with Pairwise, Triplet Ranking Losses, and ASR Pretraining
5725                                    <br>
5726                                    <span class="w3-text w3-text-theme">
5727                                        Bao Thang Ta, Minh Tu Le, Van Hai Do, Huynh Thi Thanh Binh
5728                                    </span>
5729                                </p>
5730                            </a>
5731                        </div>
5732                    </div>
5733                    <br>
5734                    <div class="w3-content" style="height:10px"  id="Privacy and Security in Speech Communication 1"></div>
5735                    <div class="w3-card w3-round w3-white w3-padding">
5736                        <div class="w3-container"  style="margin-top:40px">
5737                            <h4 class="w3-center">Privacy and Security in Speech Communication 1</h4>
5738                            <hr>
5739                            <a class="w3-text" href="muller24b_interspeech.html">
5740                                <p>
5741                                    Harder or Different? Understanding Generalization of Audio Deepfake Detection
5742                                    <br>
5743                                    <span class="w3-text w3-text-theme">
5744                                        Nicolas M. Müller, Nicholas Evans, Hemlata Tak, Philip Sperl, Konstantin Böttinger
5745                                    </span>
5746                                </p>
5747                            </a>
5748                            <a class="w3-text" href="oiso24_interspeech.html">
5749                                <p>
5750                                    Prompt Tuning for Audio Deepfake Detection: Computationally Efficient Test-time Domain Adaptation with Limited Target Dataset
5751                                    <br>
5752                                    <span class="w3-text w3-text-theme">
5753                                        Hideyuki Oiso, Yuto Matsunaga, Kazuya Kakizaki, Taiki Miyagawa
5754                                    </span>
5755                                </p>
5756                            </a>
5757                            <a class="w3-text" href="looney24_interspeech.html">
5758                                <p>
5759                                    Robust spread spectrum speech watermarking using linear prediction and deep spectral shaping
5760                                    <br>
5761                                    <span class="w3-text w3-text-theme">
5762                                        David Looney, Nikolay D. Gaubitch
5763                                    </span>
5764                                </p>
5765                            </a>
5766                            <a class="w3-text" href="chen24k_interspeech.html">
5767                                <p>
5768                                    RawBMamba: End-to-End Bidirectional State Space Model for Audio Deepfake Detection
5769                                    <br>
5770                                    <span class="w3-text w3-text-theme">
5771                                        Yujie Chen, Jiangyan Yi, Jun Xue, Chenglong Wang, Xiaohui Zhang, Shunbo Dong, Siding Zeng, Jianhua Tao, Zhao Lv, Cunhang Fan
5772                                    </span>
5773                                </p>
5774                            </a>
5775                            <a class="w3-text" href="liu24i_interspeech.html">
5776                                <p>
5777                                    How Private is Low-Frequency Speech Audio in the Wild? An Analysis of Verbal Intelligibility by Humans and Machines
5778                                    <br>
5779                                    <span class="w3-text w3-text-theme">
5780                                        Ailin Liu, Pepijn Vunderink, Jose Vargas Quiros, Chirag Raman, Hayley Hung
5781                                    </span>
5782                                </p>
5783                            </a>
5784                            <a class="w3-text" href="yang24e_interspeech.html">
5785                                <p>
5786                                    RW-VoiceShield: Raw Waveform-based Adversarial Attack on One-shot Voice Conversion
5787                                    <br>
5788                                    <span class="w3-text w3-text-theme">
5789                                        Ching-Yu Yang, Shreya G. Upadhyay, Ya-Tse Wu, Bo-Hao Su, Chi-Chun Lee
5790                                    </span>
5791                                </p>
5792                            </a>
5793                        </div>
5794                    </div>
5795                    <br>
5796                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Voice Conversion 2"></div>
5797                    <div class="w3-card w3-round w3-white w3-padding">
5798                        <div class="w3-container"  style="margin-top:40px">
5799                            <h4 class="w3-center">Speech Synthesis: Voice Conversion 2</h4>
5800                            <hr>
5801                            <a class="w3-text" href="gusev24_interspeech.html">
5802                                <p>
5803                                    Improvement Speaker Similarity for Zero-Shot Any-to-Any Voice Conversion of Whispered and Regular Speech
5804                                    <br>
5805                                    <span class="w3-text w3-text-theme">
5806                                        Aleksei Gusev, Anastasia Avdeeva
5807                                    </span>
5808                                </p>
5809                            </a>
5810                            <a class="w3-text" href="um24b_interspeech.html">
5811                                <p>
5812                                    Utilizing Adaptive Global Response Normalization and Cluster-Based Pseudo Labels for Zero-Shot Voice Conversion
5813                                    <br>
5814                                    <span class="w3-text w3-text-theme">
5815                                        Ji Sub Um, Hoirin Kim
5816                                    </span>
5817                                </p>
5818                            </a>
5819                            <a class="w3-text" href="ma24e_interspeech.html">
5820                                <p>
5821                                    Vec-Tok-VC+: Residual-enhanced Robust Zero-shot Voice Conversion with Progressive Constraints in a Dual-mode Training Strategy
5822                                    <br>
5823                                    <span class="w3-text w3-text-theme">
5824                                        Linhan Ma, Xinfa Zhu, Yuanjun Lv, Zhichao Wang, Ziqian Wang, Wendi He, Hongbin Zhou, Lei Xie
5825                                    </span>
5826                                </p>
5827                            </a>
5828                            <a class="w3-text" href="igarashi24_interspeech.html">
5829                                <p>
5830                                    Noise-Robust Voice Conversion by Conditional Denoising Training Using Latent Variables of Recording Quality and Environment
5831                                    <br>
5832                                    <span class="w3-text w3-text-theme">
5833                                        Takuto Igarashi, Yuki Saito, Kentaro Seki, Shinnosuke Takamichi, Ryuichi Yamamoto, Kentaro Tachibana, Hiroshi Saruwatari
5834                                    </span>
5835                                </p>
5836                            </a>
5837                            <a class="w3-text" href="kanagawa24_interspeech.html">
5838                                <p>
5839                                    Pre-training Neural Transducer-based Streaming Voice Conversion for Faster Convergence and Alignment-free Training
5840                                    <br>
5841                                    <span class="w3-text w3-text-theme">
5842                                        Hiroki Kanagawa, Takafumi Moriya, Yusuke Ijima
5843                                    </span>
5844                                </p>
5845                            </a>
5846                            <a class="w3-text" href="xu24b_interspeech.html">
5847                                <p>
5848                                    Residual Speaker Representation for One-Shot Voice Conversion
5849                                    <br>
5850                                    <span class="w3-text w3-text-theme">
5851                                        Le Xu, Jiangyan Yi, Tao Wang, Yong Ren, Rongxiu Zhong, Zhengqi Wen, Jianhua Tao
5852                                    </span>
5853                                </p>
5854                            </a>
5855                            <a class="w3-text" href="gengembre24_interspeech.html">
5856                                <p>
5857                                    Disentangling prosody and timbre embeddings via voice conversion
5858                                    <br>
5859                                    <span class="w3-text w3-text-theme">
5860                                        Nicolas Gengembre, Olivier Le Blouch, Cédric Gendrot
5861                                    </span>
5862                                </p>
5863                            </a>
5864                            <a class="w3-text" href="chen24e_interspeech.html">
5865                                <p>
5866                                    LDM-SVC: Latent Diffusion Model Based Zero-Shot Any-to-Any Singing Voice Conversion with Singer Guidance
5867                                    <br>
5868                                    <span class="w3-text w3-text-theme">
5869                                        Shihao Chen, Yu Gu, Jie Zhang, Na Li, Rilin Chen, Liping Chen, Lirong Dai
5870                                    </span>
5871                                </p>
5872                            </a>
5873                        </div>
5874                    </div>
5875                    <br>
5876                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Text Processing"></div>
5877                    <div class="w3-card w3-round w3-white w3-padding">
5878                        <div class="w3-container"  style="margin-top:40px">
5879                            <h4 class="w3-center">Speech Synthesis: Text Processing</h4>
5880                            <hr>
5881                            <a class="w3-text" href="roth24_interspeech.html">
5882                                <p>
5883                                    A Language Modeling Approach to Diacritic-Free Hebrew TTS
5884                                    <br>
5885                                    <span class="w3-text w3-text-theme">
5886                                        Amit Roth, Arnon Turetzky, Yossi Adi
5887                                    </span>
5888                                </p>
5889                            </a>
5890                            <a class="w3-text" href="dekel24_interspeech.html">
5891                                <p>
5892                                    Exploring the Benefits of Tokenization of Discrete Acoustic Units
5893                                    <br>
5894                                    <span class="w3-text w3-text-theme">
5895                                        Avihu Dekel, Raul Fernandez
5896                                    </span>
5897                                </p>
5898                            </a>
5899                            <a class="w3-text" href="rezackova24_interspeech.html">
5900                                <p>
5901                                    Homograph Disambiguation with Text-to-Text Transfer Transformer
5902                                    <br>
5903                                    <span class="w3-text w3-text-theme">
5904                                        Markéta Řezáčková, Daniel Tihelka, Jindřich Matoušek
5905                                    </span>
5906                                </p>
5907                            </a>
5908                            <a class="w3-text" href="kurihara24_interspeech.html">
5909                                <p>
5910                                    Enhancing Japanese Text-to-Speech Accuracy with a Novel Combination Transformer-BERT-based G2P: Integrating Pronunciation Dictionaries and Accent Sandhi
5911                                    <br>
5912                                    <span class="w3-text w3-text-theme">
5913                                        Kiyoshi Kurihara, Masanori Sano
5914                                    </span>
5915                                </p>
5916                            </a>
5917                            <a class="w3-text" href="shirahata24_interspeech.html">
5918                                <p>
5919                                    Audio-conditioned phonemic and prosodic annotation for building text-to-speech models from unlabeled speech data
5920                                    <br>
5921                                    <span class="w3-text w3-text-theme">
5922                                        Yuma Shirahata, Byeongseon Park, Ryuichi Yamamoto, Kentaro Tachibana
5923                                    </span>
5924                                </p>
5925                            </a>
5926                            <a class="w3-text" href="yang24_interspeech.html">
5927                                <p>
5928                                    G2PA: G2P with Aligned Audio for Mandarin Chinese
5929                                    <br>
5930                                    <span class="w3-text w3-text-theme">
5931                                        Xingxing Yang
5932                                    </span>
5933                                </p>
5934                            </a>
5935                            <a class="w3-text" href="sun24_interspeech.html">
5936                                <p>
5937                                    Learning Pronunciation from Other Accents via Pronunciation Knowledge Transfer
5938                                    <br>
5939                                    <span class="w3-text w3-text-theme">
5940                                        Siqi Sun, Korin Richmond
5941                                    </span>
5942                                </p>
5943                            </a>
5944                            <a class="w3-text" href="gupta24d_interspeech.html">
5945                                <p>
5946                                    Positional Description for Numerical Normalization 
5947                                    <br>
5948                                    <span class="w3-text w3-text-theme">
5949                                        Deepanshu Gupta, Javier Latorre
5950                                    </span>
5951                                </p>
5952                            </a>
5953                            <a class="w3-text" href="tannander24_interspeech.html">
5954                                <p>
5955                                    Beyond graphemes and phonemes: continuous phonological features in neural text-to-speech synthesis
5956                                    <br>
5957                                    <span class="w3-text w3-text-theme">
5958                                        Christina TÃ¥nnander, Shivam Mehta, Jonas Beskow, Jens Edlund
5959                                    </span>
5960                                </p>
5961                            </a>
5962                        </div>
5963                    </div>
5964                    <br>
5965                    <div class="w3-content" style="height:10px"  id="Training Methods, Self-Supervised Learning, Adaptation"></div>
5966                    <div class="w3-card w3-round w3-white w3-padding">
5967                        <div class="w3-container"  style="margin-top:40px">
5968                            <h4 class="w3-center">Training Methods, Self-Supervised Learning, Adaptation</h4>
5969                            <hr>
5970                            <a class="w3-text" href="fernandezlopez24_interspeech.html">
5971                                <p>
5972                                    MSRS: Training Multimodal Speech Recognition Models from Scratch with Sparse Mask Optimization
5973                                    <br>
5974                                    <span class="w3-text w3-text-theme">
5975                                        Adriana Fernandez-Lopez, Honglie Chen, Pingchuan Ma, Lu Yin, Qiao Xiao, Stavros Petridis, Shiwei Liu, Maja Pantic
5976                                    </span>
5977                                </p>
5978                            </a>
5979                            <a class="w3-text" href="prasad24_interspeech.html">
5980                                <p>
5981                                    Speech and Language Recognition with Low-rank Adaptation of Pretrained Models
5982                                    <br>
5983                                    <span class="w3-text w3-text-theme">
5984                                        Amrutha Prasad, Srikanth Madikeri, Driss Khalil, Petr Motlicek, Christof Schuepbach
5985                                    </span>
5986                                </p>
5987                            </a>
5988                            <a class="w3-text" href="kim24s_interspeech.html">
5989                                <p>
5990                                    Convolution-Augmented Parameter-Efficient Fine-Tuning for Speech Recognition
5991                                    <br>
5992                                    <span class="w3-text w3-text-theme">
5993                                        Kwangyoun Kim, Suwon Shon, Yi-Te Hsu, Prashant Sridhar, Karen Livescu, Shinji Watanabe
5994                                    </span>
5995                                </p>
5996                            </a>
5997                            <a class="w3-text" href="meghanani24_interspeech.html">
5998                                <p>
5999                                    LASER: Learning by Aligning Self-supervised Representations of Speech for Improving Content-related Tasks
6000                                    <br>
6001                                    <span class="w3-text w3-text-theme">
6002                                        Amit Meghanani, Thomas Hain
6003                                    </span>
6004                                </p>
6005                            </a>
6006                            <a class="w3-text" href="flynn24_interspeech.html">
6007                                <p>
6008                                    Self-Train Before You Transcribe
6009                                    <br>
6010                                    <span class="w3-text w3-text-theme">
6011                                        Robert Flynn, Anton Ragni
6012                                    </span>
6013                                </p>
6014                            </a>
6015                            <a class="w3-text" href="vandereeckt24_interspeech.html">
6016                                <p>
6017                                    Unsupervised Online Continual Learning for Automatic Speech Recognition
6018                                    <br>
6019                                    <span class="w3-text w3-text-theme">
6020                                        Steven Vander Eeckt, Hugo Van hamme
6021                                    </span>
6022                                </p>
6023                            </a>
6024                            <a class="w3-text" href="shi24b_interspeech.html">
6025                                <p>
6026                                    Dual-path Adaptation of Pretrained Feature Extraction Module for Robust Automatic Speech Recognition
6027                                    <br>
6028                                    <span class="w3-text w3-text-theme">
6029                                        Hao Shi, Tatsuya Kawahara
6030                                    </span>
6031                                </p>
6032                            </a>
6033                            <a class="w3-text" href="kusunoki24_interspeech.html">
6034                                <p>
6035                                    Hierarchical Multi-Task Learning with CTC and Recursive Operation
6036                                    <br>
6037                                    <span class="w3-text w3-text-theme">
6038                                        Nahomi Kusunoki, Yosuke Higuchi, Tetsuji Ogawa, Tetsunori Kobayashi
6039                                    </span>
6040                                </p>
6041                            </a>
6042                            <a class="w3-text" href="hojo24_interspeech.html">
6043                                <p>
6044                                    Boosting CTC-based ASR using inter-layer attention-based CTC loss
6045                                    <br>
6046                                    <span class="w3-text w3-text-theme">
6047                                        Keigo Hojo, Yukoh Wakabayashi, Kengo Ohta, Atsunori Ogawa, Norihide Kitaoka
6048                                    </span>
6049                                </p>
6050                            </a>
6051                            <a class="w3-text" href="kim24t_interspeech.html">
6052                                <p>
6053                                    Self-training ASR Guided by Unsupervised ASR Teacher
6054                                    <br>
6055                                    <span class="w3-text w3-text-theme">
6056                                        Hyung Yong Kim, Byeong-Yeol Kim, Yunkyu Lim, Jihwan Park, Shukjae Choi, Yooncheol Ju, Jinseok Park, Youshin Lim, Seung Woo Yu, Hanbin Lee, Shinji Watanabe
6057                                    </span>
6058                                </p>
6059                            </a>
6060                            <a class="w3-text" href="gu24b_interspeech.html">
6061                                <p>
6062                                    Personality-memory Gated Adaptation: An Efficient Speaker Adaptation for Personalized End-to-end Automatic Speech Recognition
6063                                    <br>
6064                                    <span class="w3-text w3-text-theme">
6065                                        Yue Gu, Zhihao Du, Shiliang Zhang, jiqing Han, Yongjun He
6066                                    </span>
6067                                </p>
6068                            </a>
6069                            <a class="w3-text" href="joseph24_interspeech.html">
6070                                <p>
6071                                    Speaker Personalization for Automatic Speech Recognition using Weight-Decomposed Low-Rank Adaptation
6072                                    <br>
6073                                    <span class="w3-text w3-text-theme">
6074                                        George Joseph, Arun Baby
6075                                    </span>
6076                                </p>
6077                            </a>
6078                            <a class="w3-text" href="lee24j_interspeech.html">
6079                                <p>
6080                                    Online Subloop Search via Uncertainty Quantization for Efficient Test-Time Adaptation
6081                                    <br>
6082                                    <span class="w3-text w3-text-theme">
6083                                        Jae-Hong Lee, Sang-Eon Lee, Dong-Hyun Kim, DoHee Kim, Joon-Hyuk Chang
6084                                    </span>
6085                                </p>
6086                            </a>
6087                            <a class="w3-text" href="singh24c_interspeech.html">
6088                                <p>
6089                                    ROAR: Reinforcing Original to Augmented Data Ratio Dynamics for Wav2vec2.0 Based ASR
6090                                    <br>
6091                                    <span class="w3-text w3-text-theme">
6092                                        Vishwanath Pratap Singh, Federico Malato, Ville Hautamäki, Md. Sahidullah, Tomi Kinnunen
6093                                    </span>
6094                                </p>
6095                            </a>
6096                            <a class="w3-text" href="lee24b_interspeech.html">
6097                                <p>
6098                                    Online Knowledge Distillation of Decoder-Only Large Language Models for Efficient Speech Recognition
6099                                    <br>
6100                                    <span class="w3-text w3-text-theme">
6101                                        Jeehye Lee, Hyeji Seo
6102                                    </span>
6103                                </p>
6104                            </a>
6105                        </div>
6106                    </div>
6107                    <br>
6108                    <div class="w3-content" style="height:10px"  id="Novel Architectures for ASR"></div>
6109                    <div class="w3-card w3-round w3-white w3-padding">
6110                        <div class="w3-container"  style="margin-top:40px">
6111                            <h4 class="w3-center">Novel Architectures for ASR</h4>
6112                            <hr>
6113                            <a class="w3-text" href="honda24_interspeech.html">
6114                                <p>
6115                                    Efficient and Robust Long-Form Speech Recognition with Hybrid H3-Conformer
6116                                    <br>
6117                                    <span class="w3-text w3-text-theme">
6118                                        Tomoki Honda, Shinsuke Sakai, Tatsuya Kawahara
6119                                    </span>
6120                                </p>
6121                            </a>
6122                            <a class="w3-text" href="kashiwagi24_interspeech.html">
6123                                <p>
6124                                    Rapid Language Adaptation for Multilingual E2E Speech Recognition Using Encoder Prompting
6125                                    <br>
6126                                    <span class="w3-text w3-text-theme">
6127                                        Yosuke Kashiwagi, Hayato Futami, Emiru Tsunoo, Siddhant Arora, Shinji Watanabe
6128                                    </span>
6129                                </p>
6130                            </a>
6131                            <a class="w3-text" href="shejwalkar24_interspeech.html">
6132                                <p>
6133                                    Quantifying Unintended Memorization in BEST-RQ ASR Encoders
6134                                    <br>
6135                                    <span class="w3-text w3-text-theme">
6136                                        Virat Shejwalkar, Om Thakkar, Arun Narayanan
6137                                    </span>
6138                                </p>
6139                            </a>
6140                            <a class="w3-text" href="kang24b_interspeech.html">
6141                                <p>
6142                                    SWAN: SubWord Alignment Network for HMM-free word timing estimation in end-to-end automatic speech recognition
6143                                    <br>
6144                                    <span class="w3-text w3-text-theme">
6145                                        Woo Hyun Kang, Srikanth Vishnubhotla, Rudolf Braun, Yogesh Virkar, Raghuveer Peri, Kyu J. Han
6146                                    </span>
6147                                </p>
6148                            </a>
6149                        </div>
6150                    </div>
6151                    <br>
6152                    <div class="w3-content" style="height:10px"  id="Multimodality and Foundation Models"></div>
6153                    <div class="w3-card w3-round w3-white w3-padding">
6154                        <div class="w3-container"  style="margin-top:40px">
6155                            <h4 class="w3-center">Multimodality and Foundation Models</h4>
6156                            <hr>
6157                            <a class="w3-text" href="cui24_interspeech.html">
6158                                <p>
6159                                    Spontaneous Speech-Based Suicide Risk Detection Using Whisper and Large Language Models
6160                                    <br>
6161                                    <span class="w3-text w3-text-theme">
6162                                        Ziyun Cui, Chang Lei, Wen Wu, Yinan Duan, Diyang Qu, Ji Wu, Runsen Chen, Chao Zhang
6163                                    </span>
6164                                </p>
6165                            </a>
6166                            <a class="w3-text" href="sayeed24_interspeech.html">
6167                                <p>
6168                                    Spoken Word2Vec: Learning Skipgram Embeddings from Speech
6169                                    <br>
6170                                    <span class="w3-text w3-text-theme">
6171                                        Mohammad Amaan Sayeed, Hanan Aldarmaki
6172                                    </span>
6173                                </p>
6174                            </a>
6175                            <a class="w3-text" href="bujnowski24_interspeech.html">
6176                                <p>
6177                                    SAMSEMO: New dataset for multilingual and multimodal emotion recognition
6178                                    <br>
6179                                    <span class="w3-text w3-text-theme">
6180                                        Pawel Bujnowski, Bartlomiej Kuzma, Bartlomiej Paziewski, Jacek Rutkowski, Joanna Marhula, Zuzanna Bordzicka, Piotr Andruszkiewicz
6181                                    </span>
6182                                </p>
6183                            </a>
6184                            <a class="w3-text" href="jia24_interspeech.html">
6185                                <p>
6186                                    LLM-Driven Multimodal Opinion Expression Identification
6187                                    <br>
6188                                    <span class="w3-text w3-text-theme">
6189                                        Bonian Jia, Huiyao Chen, Yueheng Sun, Meishan Zhang, Min Zhang
6190                                    </span>
6191                                </p>
6192                            </a>
6193                            <a class="w3-text" href="li24ta_interspeech.html">
6194                                <p>
6195                                    Zero-Shot Fake Video Detection by Audio-Visual Consistency
6196                                    <br>
6197                                    <span class="w3-text w3-text-theme">
6198                                        Xiaolou Li, Zehua Liu, Chen Chen, Lantian Li, Li Guo, Dong Wang
6199                                    </span>
6200                                </p>
6201                            </a>
6202                            <a class="w3-text" href="eungi24_interspeech.html">
6203                                <p>
6204                                    Enhancing Speech-Driven 3D Facial Animation with Audio-Visual Guidance from Lip Reading Expert
6205                                    <br>
6206                                    <span class="w3-text w3-text-theme">
6207                                        Han EunGi, Oh Hyun-Bin, Kim Sung-Bin, Corentin Nivelet Etcheberry, Suekyeong Nam, Janghoon Ju, Tae-Hyun Oh
6208                                    </span>
6209                                </p>
6210                            </a>
6211                        </div>
6212                    </div>
6213                    <br>
6214                    <div class="w3-content" style="height:10px"  id="Spoken Dialogue Systems and Conversational Analysis 1"></div>
6215                    <div class="w3-card w3-round w3-white w3-padding">
6216                        <div class="w3-container"  style="margin-top:40px">
6217                            <h4 class="w3-center">Spoken Dialogue Systems and Conversational Analysis 1</h4>
6218                            <hr>
6219                            <a class="w3-text" href="mcneill24_interspeech.html">
6220                                <p>
6221                                    Autoregressive cross-interlocutor attention scores meaningfully capture conversational dynamics
6222                                    <br>
6223                                    <span class="w3-text w3-text-theme">
6224                                        Matthew McNeill, Rivka Levitan
6225                                    </span>
6226                                </p>
6227                            </a>
6228                            <a class="w3-text" href="atkins24_interspeech.html">
6229                                <p>
6230                                    ConvoCache: Smart Re-Use of Chatbot Responses
6231                                    <br>
6232                                    <span class="w3-text w3-text-theme">
6233                                        Conor Atkins, Ian Wood, Mohamed Ali Kaafar, Hassan Asghar, Nardine Basta, Michal Kepkowski
6234                                    </span>
6235                                </p>
6236                            </a>
6237                            <a class="w3-text" href="qian24b_interspeech.html">
6238                                <p>
6239                                    Joint Learning of Context and Feedback Embeddings in Spoken Dialogue
6240                                    <br>
6241                                    <span class="w3-text w3-text-theme">
6242                                        Livia Qian, Gabriel Skantze
6243                                    </span>
6244                                </p>
6245                            </a>
6246                            <a class="w3-text" href="sahipjohn24_interspeech.html">
6247                                <p>
6248                                    DubWise: Video-Guided Speech Duration Control in Multimodal LLM-based Text-to-Speech for Dubbing
6249                                    <br>
6250                                    <span class="w3-text w3-text-theme">
6251                                        Neha Sahipjohn, Ashishkumar Gudmalwar, Nirmesh Shah, Pankaj Wasnik, Rajiv Ratn Shah
6252                                    </span>
6253                                </p>
6254                            </a>
6255                            <a class="w3-text" href="wang24t_interspeech.html">
6256                                <p>
6257                                    Contextual Interactive Evaluation of TTS Models in Dialogue Systems
6258                                    <br>
6259                                    <span class="w3-text w3-text-theme">
6260                                        Siyang Wang, Éva Székely, Joakim Gustafson
6261                                    </span>
6262                                </p>
6263                            </a>
6264                            <a class="w3-text" href="shih24b_interspeech.html">
6265                                <p>
6266                                    GSQA: An End-to-End Model for Generative Spoken Question Answering
6267                                    <br>
6268                                    <span class="w3-text w3-text-theme">
6269                                        Min-Han Shih, Ho-Lam Chung, Yu-Chi Pai, Ming-Hao Hsu, Guan-Ting Lin, Shang-Wen Li, Hung-yi Lee
6270                                    </span>
6271                                </p>
6272                            </a>
6273                        </div>
6274                    </div>
6275                    <br>
6276                    <div class="w3-content" style="height:10px"  id="Speech Technology"></div>
6277                    <div class="w3-card w3-round w3-white w3-padding">
6278                        <div class="w3-container"  style="margin-top:40px">
6279                            <h4 class="w3-center">Speech Technology</h4>
6280                            <hr>
6281                            <a class="w3-text" href="nilsson24_interspeech.html">
6282                                <p>
6283                                    Resource-Efficient Speech Quality Prediction through Quantization Aware Training and Binary Activation Maps
6284                                    <br>
6285                                    <span class="w3-text w3-text-theme">
6286                                        Mattias Nilsson, Riccardo Miccini, Clement Laroche, Tobias Piechowiak, Friedemann Zenke
6287                                    </span>
6288                                </p>
6289                            </a>
6290                            <a class="w3-text" href="naderi24_interspeech.html">
6291                                <p>
6292                                    Towards interfacing large language models with ASR systems using confidence measures and prompting
6293                                    <br>
6294                                    <span class="w3-text w3-text-theme">
6295                                        Maryam Naderi, Enno Hermann, Alexandre Nanchen, Sevada Hovsepyan, Mathew Magimai.-Doss
6296                                    </span>
6297                                </p>
6298                            </a>
6299                            <a class="w3-text" href="meng24d_interspeech.html">
6300                                <p>
6301                                    Text Injection for Neural Contextual Biasing
6302                                    <br>
6303                                    <span class="w3-text w3-text-theme">
6304                                        Zhong Meng, Zelin Wu, Rohit Prabhavalkar, Cal Peyser, Weiran Wang, Nanxin Chen, Tara N. Sainath, Bhuvana Ramabhadran
6305                                    </span>
6306                                </p>
6307                            </a>
6308                            <a class="w3-text" href="wu24l_interspeech.html">
6309                                <p>
6310                                    Prompting Large Language Models with Mispronunciation Detection and Diagnosis Abilities
6311                                    <br>
6312                                    <span class="w3-text w3-text-theme">
6313                                        Minglin Wu, Jing Xu, Xixin Wu, Helen Meng
6314                                    </span>
6315                                </p>
6316                            </a>
6317                            <a class="w3-text" href="sun24d_interspeech.html">
6318                                <p>
6319                                    Acceleration of Posteriorgram-based DTW by Distilling the Class-to-class Distances Encoded in the Classifier Used to Calculate Posteriors
6320                                    <br>
6321                                    <span class="w3-text w3-text-theme">
6322                                        Haitong Sun, Jaehyun Choi, Nobuaki Minematsu, Daisuke Saito
6323                                    </span>
6324                                </p>
6325                            </a>
6326                            <a class="w3-text" href="gudmalwar24_interspeech.html">
6327                                <p>
6328                                    VECL-TTS: Voice identity and Emotional style controllable Cross-Lingual Text-to-Speech
6329                                    <br>
6330                                    <span class="w3-text w3-text-theme">
6331                                        Ashishkumar Gudmalwar, Nirmesh Shah, Sai Akarsh, Pankaj Wasnik, Rajiv Ratn Shah
6332                                    </span>
6333                                </p>
6334                            </a>
6335                            <a class="w3-text" href="svirsky24_interspeech.html">
6336                                <p>
6337                                    Sparse Binarization for Fast Keyword Spotting
6338                                    <br>
6339                                    <span class="w3-text w3-text-theme">
6340                                        Jonathan Svirsky, Uri Shaham, Ofir Lindenbaum
6341                                    </span>
6342                                </p>
6343                            </a>
6344                        </div>
6345                    </div>
6346                    <br>
6347                    <div class="w3-content" style="height:10px"  id="Pathological Speech Analysis 2"></div>
6348                    <div class="w3-card w3-round w3-white w3-padding">
6349                        <div class="w3-container"  style="margin-top:40px">
6350                            <h4 class="w3-center">Pathological Speech Analysis 2</h4>
6351                            <hr>
6352                            <a class="w3-text" href="halpern24_interspeech.html">
6353                                <p>
6354                                    Quantifying the effect of speech pathology on automatic and human speaker verification
6355                                    <br>
6356                                    <span class="w3-text w3-text-theme">
6357                                        Bence Mark Halpern, Thomas Tienkamp, Wen-Chin Huang, Lester Phillip Violeta, Teja Rebernik, Sebastiaan de Visscher, Max Witjes, Martijn Wieling, Defne Abur, Tomoki Toda
6358                                    </span>
6359                                </p>
6360                            </a>
6361                            <a class="w3-text" href="maji24_interspeech.html">
6362                                <p>
6363                                    Investigation of Layer-Wise Speech Representations in Self-Supervised Learning Models: A Cross-Lingual Study in Detecting Depression
6364                                    <br>
6365                                    <span class="w3-text w3-text-theme">
6366                                        Bubai Maji, Rajlakshmi Guha, Aurobinda Routray, Shazia Nasreen, Debabrata Majumdar
6367                                    </span>
6368                                </p>
6369                            </a>
6370                            <a class="w3-text" href="talkar24_interspeech.html">
6371                                <p>
6372                                    Detection of Cognitive Impairment And Alzheimer's Disease Using a Speech- and Language-Based Protocol
6373                                    <br>
6374                                    <span class="w3-text w3-text-theme">
6375                                        Tanya Talkar, Sherman Charles, Chelsea Krantsevich, Kan Kawabata
6376                                    </span>
6377                                </p>
6378                            </a>
6379                            <a class="w3-text" href="lin24l_interspeech.html">
6380                                <p>
6381                                    Analyzing Multimodal Features of Spontaneous Voice Assistant Commands for Mild Cognitive Impairment Detection
6382                                    <br>
6383                                    <span class="w3-text w3-text-theme">
6384                                        Nana Lin, Youxiang Zhu, Xiaohui Liang, John A. Batsis, Caroline Summerour
6385                                    </span>
6386                                </p>
6387                            </a>
6388                            <a class="w3-text" href="woszczyk24_interspeech.html">
6389                                <p>
6390                                    Prosody-Driven Privacy-Preserving Dementia Detection
6391                                    <br>
6392                                    <span class="w3-text w3-text-theme">
6393                                        Dominika Woszczyk, Ranya Aloufi, Soteris Demetriou
6394                                    </span>
6395                                </p>
6396                            </a>
6397                            <a class="w3-text" href="koudounas24_interspeech.html">
6398                                <p>
6399                                    Voice Disorder Analysis: a Transformer-based Approach
6400                                    <br>
6401                                    <span class="w3-text w3-text-theme">
6402                                        Alkis Koudounas, Gabriele Ciravegna, Marco Fantini, Erika Crosetti, Giovanni Succo, Tania Cerquitelli, Elena Baralis
6403                                    </span>
6404                                </p>
6405                            </a>
6406                        </div>
6407                    </div>
6408                    <br>
6409                    <div class="w3-content" style="height:10px"  id="Speech Science, Speech Technology, and Gender (Special Session)"></div>
6410                    <div class="w3-card w3-round w3-white w3-padding">
6411                        <div class="w3-container"  style="margin-top:40px">
6412                            <h4 class="w3-center">Speech Science, Speech Technology, and Gender (Special Session)</h4>
6413                            <hr>
6414                            <a class="w3-text" href="schubert24_interspeech.html">
6415                                <p>
6416                                    Challenges of German Speech Recognition: A Study on Multi-ethnolectal Speech Among Adolescents
6417                                    <br>
6418                                    <span class="w3-text w3-text-theme">
6419                                        Martha Schubert, Daniel Duran, Ingo Siegert
6420                                    </span>
6421                                </p>
6422                            </a>
6423                            <a class="w3-text" href="sigurgeirsson24_interspeech.html">
6424                                <p>
6425                                    Just Because We Camp, Doesn't Mean We Should: The Ethics of Modelling Queer Voices.
6426                                    <br>
6427                                    <span class="w3-text w3-text-theme">
6428                                        Atli Sigurgeirsson, Eddie L. Ungless
6429                                    </span>
6430                                </p>
6431                            </a>
6432                            <a class="w3-text" href="pelloin24_interspeech.html">
6433                                <p>
6434                                    Automatic Classification of News Subjects in Broadcast News: Application to a Gender Bias Representation Analysis
6435                                    <br>
6436                                    <span class="w3-text w3-text-theme">
6437                                        Valentin Pelloin, Léna Dodson, Émile Chapuis, Nicolas Hervé, David Doukhan
6438                                    </span>
6439                                </p>
6440                            </a>
6441                            <a class="w3-text" href="doukhan24_interspeech.html">
6442                                <p>
6443                                    Gender Representation in TV and Radio: Automatic Information Extraction methods versus Manual Analyses
6444                                    <br>
6445                                    <span class="w3-text w3-text-theme">
6446                                        David Doukhan, Lena Dodson, Manon Conan, Valentin Pelloin, Aurélien Clamouse, Mélina Lepape, Géraldine Van Hille, Cécile Méadel, Marlène Coulomb-Gully
6447                                    </span>
6448                                </p>
6449                            </a>
6450                            <a class="w3-text" href="hughes24_interspeech.html">
6451                                <p>
6452                                    Acoustic Effects of Facial Feminisation Surgery on Speech and Singing: A Case Study
6453                                    <br>
6454                                    <span class="w3-text w3-text-theme">
6455                                        Cliodhna Hughes, Guy Brown, Ning Ma, Nicola Dibben
6456                                    </span>
6457                                </p>
6458                            </a>
6459                            <a class="w3-text" href="szekely24_interspeech.html">
6460                                <p>
6461                                    An inclusive approach to creating a palette of synthetic voices for gender diversity
6462                                    <br>
6463                                    <span class="w3-text w3-text-theme">
6464                                        Eva Szekely, Maxwell Hope
6465                                    </span>
6466                                </p>
6467                            </a>
6468                            <a class="w3-text" href="netzorg24_interspeech.html">
6469                                <p>
6470                                    Speech After Gender: A Trans-Feminine Perspective on Next Steps for Speech Science and Technology
6471                                    <br>
6472                                    <span class="w3-text w3-text-theme">
6473                                        Robin Netzorg, Alyssa Cote, Sumi Koshin, Klo Vivienne Garoute, Gopala Krishna Anumanchipalli
6474                                    </span>
6475                                </p>
6476                            </a>
6477                            <a class="w3-text" href="lai24_interspeech.html">
6478                                <p>
6479                                    Voice Quality Variation in AAE: An Additional Challenge for Addressing Bias in ASR Models?
6480                                    <br>
6481                                    <span class="w3-text w3-text-theme">
6482                                        Li-Fang Lai, Nicole Holliday
6483                                    </span>
6484                                </p>
6485                            </a>
6486                            <a class="w3-text" href="elie24b_interspeech.html">
6487                                <p>
6488                                    Articulatory Configurations across Genders and Periods in French Radio and TV archives
6489                                    <br>
6490                                    <span class="w3-text w3-text-theme">
6491                                        Benjamin Elie, David Doukhan, Rémi Uro, Lucas Ondel-Yang, Albert Rilliard, Simon Devauchelle
6492                                    </span>
6493                                </p>
6494                            </a>
6495                            <a class="w3-text" href="krishnan24_interspeech.html">
6496                                <p>
6497                                    On the Encoding of Gender in Transformer-based ASR Representations
6498                                    <br>
6499                                    <span class="w3-text w3-text-theme">
6500                                        Aravind Krishnan, Badr M. Abdullah, Dietrich Klakow
6501                                    </span>
6502                                </p>
6503                            </a>
6504                        </div>
6505                    </div>
6506                    <br>
6507                    <div class="w3-content" style="height:10px"  id="Speech Production and Perception"></div>
6508                    <div class="w3-card w3-round w3-white w3-padding">
6509                        <div class="w3-container"  style="margin-top:40px">
6510                            <h4 class="w3-center">Speech Production and Perception</h4>
6511                            <hr>
6512                            <a class="w3-text" href="fan24c_interspeech.html">
6513                                <p>
6514                                    Towards a Quantitative Analysis of Coarticulation with a Phoneme-to-Articulatory Model
6515                                    <br>
6516                                    <span class="w3-text w3-text-theme">
6517                                        Chaofei Fan, Jaimie M. Henderson, Chris Manning, Francis R. Willett
6518                                    </span>
6519                                </p>
6520                            </a>
6521                            <a class="w3-text" href="sharma24_interspeech.html">
6522                                <p>
6523                                    A comparative study of the impact of voiceless alveolar and palato-alveolar sibilants in English on lip aperture and protrusion during VCV production
6524                                    <br>
6525                                    <span class="w3-text w3-text-theme">
6526                                        Chetan Sharma, Vaishnavi Chandwanshi, Prasanta Kumar Ghosh
6527                                    </span>
6528                                </p>
6529                            </a>
6530                            <a class="w3-text" href="birkholz24_interspeech.html">
6531                                <p>
6532                                    Measurement and simulation of pressure losses due to airflow in vocal tract models
6533                                    <br>
6534                                    <span class="w3-text w3-text-theme">
6535                                        Peter Birkholz, Patrick Häsner
6536                                    </span>
6537                                </p>
6538                            </a>
6539                            <a class="w3-text" href="fang24_interspeech.html">
6540                                <p>
6541                                    On The Performance of EMA-synchronized Speech and Stand-alone Speech in Acoustic-to-articulatory Inversion
6542                                    <br>
6543                                    <span class="w3-text w3-text-theme">
6544                                        Qiang Fang
6545                                    </span>
6546                                </p>
6547                            </a>
6548                            <a class="w3-text" href="freixes24_interspeech.html">
6549                                <p>
6550                                    Glottal inverse filtering and vocal tract tuning for the numerical simulation of vowel /a/ with different levels of vocal effort
6551                                    <br>
6552                                    <span class="w3-text w3-text-theme">
6553                                        Marc Freixes, Marc Arnela, Joan Claudi Socoró, Luis Joglar-Ongay, Oriol Guasch, Francesc Alías-Pujol
6554                                    </span>
6555                                </p>
6556                            </a>
6557                            <a class="w3-text" href="friedrichs24_interspeech.html">
6558                                <p>
6559                                    Temporal Co-Registration of Simultaneous Electromagnetic Articulography and Electroencephalography for Precise Articulatory and Neural Data Alignment
6560                                    <br>
6561                                    <span class="w3-text w3-text-theme">
6562                                        Daniel Friedrichs, Monica Lancheros, Sam Kirkham, Lei He, Andrew Clark, Clemens Lutz, Volker Dellwo, Steven Moran
6563                                    </span>
6564                                </p>
6565                            </a>
6566                        </div>
6567                    </div>
6568                    <br>
6569                    <div class="w3-content" style="height:10px"  id="Phonetics and Phonology: Segmentals and Suprasegmentals"></div>
6570                    <div class="w3-card w3-round w3-white w3-padding">
6571                        <div class="w3-container"  style="margin-top:40px">
6572                            <h4 class="w3-center">Phonetics and Phonology: Segmentals and Suprasegmentals</h4>
6573                            <hr>
6574                            <a class="w3-text" href="miodonska24_interspeech.html">
6575                                <p>
6576                                    Frication noise features of Polish voiceless dental fricative and affricate produced by children with and without speech disorder
6577                                    <br>
6578                                    <span class="w3-text w3-text-theme">
6579                                        Zuzanna Miodonska, Michal Kręcichwost, Ewa Kwaśniok, Agata Sage, Pawel Badura
6580                                    </span>
6581                                </p>
6582                            </a>
6583                            <a class="w3-text" href="hu24_interspeech.html">
6584                                <p>
6585                                    Key Acoustic Cues for the Realization of Metrical Prominence in Tone Languages: A Cross-Dialect Study
6586                                    <br>
6587                                    <span class="w3-text w3-text-theme">
6588                                        Yiying Hu, Hui Feng
6589                                    </span>
6590                                </p>
6591                            </a>
6592                            <a class="w3-text" href="watkins24_interspeech.html">
6593                                <p>
6594                                    Revisiting Pitch Jumps: F0 Ratio in Seoul Korean
6595                                    <br>
6596                                    <span class="w3-text w3-text-theme">
6597                                        Michaela Watkins, Paul Boersma, Silke Hamann
6598                                    </span>
6599                                </p>
6600                            </a>
6601                            <a class="w3-text" href="maselli24_interspeech.html">
6602                                <p>
6603                                    Aerodynamics of Sakata labial-velar oral stops
6604                                    <br>
6605                                    <span class="w3-text w3-text-theme">
6606                                        Lorenzo Maselli, Véronique Delvaux
6607                                    </span>
6608                                </p>
6609                            </a>
6610                            <a class="w3-text" href="erickson24_interspeech.html">
6611                                <p>
6612                                    Collecting Mandible Movement in Brazilian Portuguese
6613                                    <br>
6614                                    <span class="w3-text w3-text-theme">
6615                                        Donna Erickson, Albert Rilliard, Malin Svensson Lundmark, Adelaide Silva, Leticia Rebollo Couto, Oliver Niebuhr, João Antonio de Moraes
6616                                    </span>
6617                                </p>
6618                            </a>
6619                            <a class="w3-text" href="chan24_interspeech.html">
6620                                <p>
6621                                    Pitch-driven adjustments in tongue positions: Insights from ultrasound imaging
6622                                    <br>
6623                                    <span class="w3-text w3-text-theme">
6624                                        May Pik Yu Chan, Jianjing Kuang
6625                                    </span>
6626                                </p>
6627                            </a>
6628                        </div>
6629                    </div>
6630                    <br>
6631                    <div class="w3-content" style="height:10px"  id="Topics in Paralinguistics"></div>
6632                    <div class="w3-card w3-round w3-white w3-padding">
6633                        <div class="w3-container"  style="margin-top:40px">
6634                            <h4 class="w3-center">Topics in Paralinguistics</h4>
6635                            <hr>
6636                            <a class="w3-text" href="bn24_interspeech.html">
6637                                <p>
6638                                    Speaking of Health: Leveraging Large Language Models to assess Exercise Motivation and Behavior of Rehabilitation Patients
6639                                    <br>
6640                                    <span class="w3-text w3-text-theme">
6641                                        Suhas BN, Amanda Rebar, Saeed Abdullah
6642                                    </span>
6643                                </p>
6644                            </a>
6645                            <a class="w3-text" href="wu24e_interspeech.html">
6646                                <p>
6647                                    Confidence Estimation for Automatic Detection of Depression and Alzheimer’s Disease Based on Clinical Interviews
6648                                    <br>
6649                                    <span class="w3-text w3-text-theme">
6650                                        Wen Wu, Chao Zhang, Philip C. Woodland
6651                                    </span>
6652                                </p>
6653                            </a>
6654                            <a class="w3-text" href="suda24_interspeech.html">
6655                                <p>
6656                                    Who Finds This Voice Attractive? A Large-Scale Experiment Using In-the-Wild Data
6657                                    <br>
6658                                    <span class="w3-text w3-text-theme">
6659                                        Hitoshi Suda, Aya Watanabe, Shinnosuke Takamichi
6660                                    </span>
6661                                </p>
6662                            </a>
6663                            <a class="w3-text" href="setoguchi24_interspeech.html">
6664                                <p>
6665                                    Acoustical analysis of the initial phones in speech-laugh
6666                                    <br>
6667                                    <span class="w3-text w3-text-theme">
6668                                        Ryo Setoguchi, Yoshiko Arimoto
6669                                    </span>
6670                                </p>
6671                            </a>
6672                            <a class="w3-text" href="hao24_interspeech.html">
6673                                <p>
6674                                    On Calibration of Speech Classification Models: Insights from Energy-Based Model Investigations
6675                                    <br>
6676                                    <span class="w3-text w3-text-theme">
6677                                        Yaqian Hao, Chenguang Hu, Yingying Gao, Shilei Zhang, Junlan Feng
6678                                    </span>
6679                                </p>
6680                            </a>
6681                            <a class="w3-text" href="liu24r_interspeech.html">
6682                                <p>
6683                                    Emotion-Aware Speech Self-Supervised Representation Learning with Intensity Knowledge
6684                                    <br>
6685                                    <span class="w3-text w3-text-theme">
6686                                        Rui Liu, Zening Ma
6687                                    </span>
6688                                </p>
6689                            </a>
6690                        </div>
6691                    </div>
6692                    <br>
6693                    <div class="w3-content" style="height:10px"  id="Emotion Recognition: Fairness, Variability, Uncertainty"></div>
6694                    <div class="w3-card w3-round w3-white w3-padding">
6695                        <div class="w3-container"  style="margin-top:40px">
6696                            <h4 class="w3-center">Emotion Recognition: Fairness, Variability, Uncertainty</h4>
6697                            <hr>
6698                            <a class="w3-text" href="wu24_interspeech.html">
6699                                <p>
6700                                    Dual-Constrained Dynamical Neural ODEs for Ambiguity-aware Continuous Emotion Prediction
6701                                    <br>
6702                                    <span class="w3-text w3-text-theme">
6703                                        Jingyao Wu, Ting Dang, Vidhyasaharan Sethu, Eliathamby Ambikairajah
6704                                    </span>
6705                                </p>
6706                            </a>
6707                            <a class="w3-text" href="chou24_interspeech.html">
6708                                <p>
6709                                    An Inter-Speaker Fairness-Aware Speech Emotion Regression Framework
6710                                    <br>
6711                                    <span class="w3-text w3-text-theme">
6712                                        Hsing-Hang Chou, Woan-Shiuan Chien, Ya-Tse Wu, Chi-Chun Lee
6713                                    </span>
6714                                </p>
6715                            </a>
6716                            <a class="w3-text" href="tavernor24_interspeech.html">
6717                                <p>
6718                                    The Whole Is Bigger Than the Sum of Its Parts: Modeling Individual Annotators to Capture Emotional Variability
6719                                    <br>
6720                                    <span class="w3-text w3-text-theme">
6721                                        James Tavernor, Yara El-Tawil, Emily Mower Provost
6722                                    </span>
6723                                </p>
6724                            </a>
6725                            <a class="w3-text" href="sun24e_interspeech.html">
6726                                <p>
6727                                    Iterative Prototype Refinement for Ambiguous Speech Emotion Recognition
6728                                    <br>
6729                                    <span class="w3-text w3-text-theme">
6730                                        Haoqin Sun, Shiwan Zhao, Xiangyu Kong, Xuechen Wang, Hui Wang, Jiaming Zhou, Yong Qin
6731                                    </span>
6732                                </p>
6733                            </a>
6734                            <a class="w3-text" href="chien24_interspeech.html">
6735                                <p>
6736                                    An Investigation of Group versus Individual Fairness in Perceptually Fair Speech Emotion Recognition
6737                                    <br>
6738                                    <span class="w3-text w3-text-theme">
6739                                        Woan-Shiuan Chien, Chi-Chun Lee
6740                                    </span>
6741                                </p>
6742                            </a>
6743                            <a class="w3-text" href="schrufer24_interspeech.html">
6744                                <p>
6745                                    Are you sure? Analysing Uncertainty Quantification Approaches for Real-world Speech Emotion Recognition
6746                                    <br>
6747                                    <span class="w3-text w3-text-theme">
6748                                        Oliver Schrüfer, Manuel Milling, Felix Burkhardt, Florian Eyben, Björn Schuller
6749                                    </span>
6750                                </p>
6751                            </a>
6752                            <a class="w3-text" href="garcia24_interspeech.html">
6753                                <p>
6754                                    Speech emotion recognition with deep learning beamforming  on a distant human-robot interaction scenario
6755                                    <br>
6756                                    <span class="w3-text w3-text-theme">
6757                                        Ricardo García, Rodrigo Mahu, Nicolás Grágeda, Alejandro Luzanto, Nicolas Bohmer, Carlos Busso, Néstor Becerra Yoma
6758                                    </span>
6759                                </p>
6760                            </a>
6761                        </div>
6762                    </div>
6763                    <br>
6764                    <div class="w3-content" style="height:10px"  id="Speaker Verification"></div>
6765                    <div class="w3-card w3-round w3-white w3-padding">
6766                        <div class="w3-container"  style="margin-top:40px">
6767                            <h4 class="w3-center">Speaker Verification</h4>
6768                            <hr>
6769                            <a class="w3-text" href="stafylakis24_interspeech.html">
6770                                <p>
6771                                    Challenging margin-based speaker embedding extractors by using the variational information bottleneck
6772                                    <br>
6773                                    <span class="w3-text w3-text-theme">
6774                                        Themos Stafylakis, Anna Silnova, Johan Rohdin, Oldřich Plchot, Lukáš Burget
6775                                    </span>
6776                                </p>
6777                            </a>
6778                            <a class="w3-text" href="chien24c_interspeech.html">
6779                                <p>
6780                                    Collaborative Contrastive Learning for Hypothesis Domain Adaptation
6781                                    <br>
6782                                    <span class="w3-text w3-text-theme">
6783                                        Jen-Tzung Chien, I-Ping Yeh, Man-Wai Mak
6784                                    </span>
6785                                </p>
6786                            </a>
6787                            <a class="w3-text" href="benamor24_interspeech.html">
6788                                <p>
6789                                    Extraction of interpretable and shared speaker-specific speech attributes through binary auto-encoder
6790                                    <br>
6791                                    <span class="w3-text w3-text-theme">
6792                                        Imen Ben-Amor, Jean-Francois Bonastre, Salima Mdhaffar
6793                                    </span>
6794                                </p>
6795                            </a>
6796                            <a class="w3-text" href="yakovlev24_interspeech.html">
6797                                <p>
6798                                    Reshape Dimensions Network for Speaker Recognition
6799                                    <br>
6800                                    <span class="w3-text w3-text-theme">
6801                                        Ivan Yakovlev, Rostislav Makarov, Andrei Balykin, Pavel Malov, Anton Okhotnikov, Nikita Torgashov
6802                                    </span>
6803                                </p>
6804                            </a>
6805                            <a class="w3-text" href="jung24d_interspeech.html">
6806                                <p>
6807                                    To what extent can ASV systems naturally defend against spoofing attacks?
6808                                    <br>
6809                                    <span class="w3-text w3-text-theme">
6810                                        Jee-weon Jung, Xin Wang, Nicholas Evans, Shinji Watanabe, Hye-jin Shim, Hemlata Tak, Siddhant Arora, Junichi Yamagishi, Joon Son Chung
6811                                    </span>
6812                                </p>
6813                            </a>
6814                            <a class="w3-text" href="chen24l_interspeech.html">
6815                                <p>
6816                                    ERes2NetV2: Boosting Short-Duration Speaker Verification Performance with Computational Efficiency
6817                                    <br>
6818                                    <span class="w3-text w3-text-theme">
6819                                        Yafeng Chen, Siqi Zheng, Hui Wang, Luyao Cheng, Qian Chen, Shiliang Zhang, Junjie Li
6820                                    </span>
6821                                </p>
6822                            </a>
6823                        </div>
6824                    </div>
6825                    <br>
6826                    <div class="w3-content" style="height:10px"  id="Spatial Audio and Acoustics"></div>
6827                    <div class="w3-card w3-round w3-white w3-padding">
6828                        <div class="w3-container"  style="margin-top:40px">
6829                            <h4 class="w3-center">Spatial Audio and Acoustics</h4>
6830                            <hr>
6831                            <a class="w3-text" href="khokhlov24_interspeech.html">
6832                                <p>
6833                                    Classification of Room Impulse Responses and its application for channel verification and diarization
6834                                    <br>
6835                                    <span class="w3-text w3-text-theme">
6836                                        Yuri Khokhlov, Tatiana Prisyach, Anton Mitrofanov, Dmitry Dutov, Igor Agafonov, Tatiana Timofeeva, Aleksei Romanenko, Maxim Korenevsky
6837                                    </span>
6838                                </p>
6839                            </a>
6840                            <a class="w3-text" href="kelley24_interspeech.html">
6841                                <p>
6842                                    RIR-in-a-Box: Estimating Room Acoustics from 3D Mesh Data through Shoebox Approximation
6843                                    <br>
6844                                    <span class="w3-text w3-text-theme">
6845                                        Liam Kelley, Diego Di Carlo, Aditya Arie Nugraha, Mathieu Fontaine, Yoshiaki Bando, Kazuyoshi Yoshii
6846                                    </span>
6847                                </p>
6848                            </a>
6849                            <a class="w3-text" href="ahn24c_interspeech.html">
6850                                <p>
6851                                    Novel-view Acoustic Synthesis From 3D Reconstructed Rooms
6852                                    <br>
6853                                    <span class="w3-text w3-text-theme">
6854                                        Byeongjoo Ahn, Karren Yang, Brian Hamilton, Jonathan Sheaffer, Anurag Ranjan, Miguel Sarabia, Oncel Tuzel, Jen-Hao Rick Chang
6855                                    </span>
6856                                </p>
6857                            </a>
6858                            <a class="w3-text" href="tao24_interspeech.html">
6859                                <p>
6860                                    Spatial Acoustic Enhancement Using Unbiased Relative Harmonic Coefficients
6861                                    <br>
6862                                    <span class="w3-text w3-text-theme">
6863                                        Liang Tao, Maoshen Jia, Yonggang Hu, Changchun Bao
6864                                    </span>
6865                                </p>
6866                            </a>
6867                            <a class="w3-text" href="bayestehtashk24_interspeech.html">
6868                                <p>
6869                                    Design of Feedback Active Noise Cancellation Filter Using Nested Recurrent Neural Networks
6870                                    <br>
6871                                    <span class="w3-text w3-text-theme">
6872                                        Alireza Bayestehtashk, Amit Kumar, Mike Wurtz
6873                                    </span>
6874                                </p>
6875                            </a>
6876                            <a class="w3-text" href="yarga24_interspeech.html">
6877                                <p>
6878                                    Neuromorphic Keyword Spotting with Pulse Density Modulation MEMS Microphones
6879                                    <br>
6880                                    <span class="w3-text w3-text-theme">
6881                                        Sidi Yaya Arnaud Yarga, Sean U N Wood
6882                                    </span>
6883                                </p>
6884                            </a>
6885                            <a class="w3-text" href="bitterman24_interspeech.html">
6886                                <p>
6887                                    RevRIR: Joint Reverberant Speech and Room Impulse Response Embedding using Contrastive Learning with Application to Room Shape Classification
6888                                    <br>
6889                                    <span class="w3-text w3-text-theme">
6890                                        Jacob Bitterman, Daniel Levi, Hilel Hagai Diamandi, Sharon Gannot, Tal Rosenwein
6891                                    </span>
6892                                </p>
6893                            </a>
6894                        </div>
6895                    </div>
6896                    <br>
6897                    <div class="w3-content" style="height:10px"  id="Generative Models for Speech and Audio"></div>
6898                    <div class="w3-card w3-round w3-white w3-padding">
6899                        <div class="w3-container"  style="margin-top:40px">
6900                            <h4 class="w3-center">Generative Models for Speech and Audio</h4>
6901                            <hr>
6902                            <a class="w3-text" href="bai24b_interspeech.html">
6903                                <p>
6904                                    ConsistencyTTA: Accelerating Diffusion-Based Text-to-Audio Generation with Consistency Distillation
6905                                    <br>
6906                                    <span class="w3-text w3-text-theme">
6907                                        Yatong Bai, Trung Dang, Dung Tran, Kazuhito Koishida, Somayeh Sojoudi
6908                                    </span>
6909                                </p>
6910                            </a>
6911                            <a class="w3-text" href="paissan24b_interspeech.html">
6912                                <p>
6913                                    Audio Editing with Non-Rigid Text Prompts
6914                                    <br>
6915                                    <span class="w3-text w3-text-theme">
6916                                        Francesco Paissan, Luca Della Libera, Zhepei Wang, Paris Smaragdis, Mirco Ravanelli, Cem Subakan
6917                                    </span>
6918                                </p>
6919                            </a>
6920                            <a class="w3-text" href="gupta24b_interspeech.html">
6921                                <p>
6922                                    Phoneme Discretized Saliency Maps for Explainable Detection of AI-Generated Voice
6923                                    <br>
6924                                    <span class="w3-text w3-text-theme">
6925                                        Shubham Gupta, Mirco Ravanelli, Pascal Germain, Cem Subakan
6926                                    </span>
6927                                </p>
6928                            </a>
6929                            <a class="w3-text" href="moschopoulos24_interspeech.html">
6930                                <p>
6931                                    Exploring compressibility of transformer based text-to-music (TTM) models
6932                                    <br>
6933                                    <span class="w3-text w3-text-theme">
6934                                        Vasileios Moschopoulos, Thanasis Kotsiopoulos, Pablo Peso Parada, Konstantinos Nikiforidis, Alexandros Stergiadis, Gerasimos Papakostas, Md Asif Jalal, Jisi Zhang, Anastasios Drosou, Karthikeyan Saravanan
6935                                    </span>
6936                                </p>
6937                            </a>
6938                            <a class="w3-text" href="kim24n_interspeech.html">
6939                                <p>
6940                                    Sound of Vision: Audio Generation from Visual Text Embedding through Training Domain Discriminator
6941                                    <br>
6942                                    <span class="w3-text w3-text-theme">
6943                                        Jaewon Kim, Won-Gook Choi, Seyun Ahn, Joon-Hyuk Chang
6944                                    </span>
6945                                </p>
6946                            </a>
6947                            <a class="w3-text" href="choi24c_interspeech.html">
6948                                <p>
6949                                    Retrieval-Augmented Classifier Guidance for Audio Generation
6950                                    <br>
6951                                    <span class="w3-text w3-text-theme">
6952                                        Ho-Young Choi, Won-Gook Choi, Joon-Hyuk Chang
6953                                    </span>
6954                                </p>
6955                            </a>
6956                            <a class="w3-text" href="cappellazzo24_interspeech.html">
6957                                <p>
6958                                    Efficient Fine-tuning of Audio Spectrogram Transformers via Soft Mixture of Adapters
6959                                    <br>
6960                                    <span class="w3-text w3-text-theme">
6961                                        Umberto Cappellazzo, Daniele Falavigna, Alessio Brutti
6962                                    </span>
6963                                </p>
6964                            </a>
6965                            <a class="w3-text" href="deshmukh24b_interspeech.html">
6966                                <p>
6967                                    PAM: Prompting Audio-Language Models for Audio Quality Assessment
6968                                    <br>
6969                                    <span class="w3-text w3-text-theme">
6970                                        Soham Deshmukh, Dareen Alharthi, Benjamin Elizalde, Hannes Gamper, Mahmoud Al Ismail, Rita Singh, Bhiksha Raj, Huaming Wang
6971                                    </span>
6972                                </p>
6973                            </a>
6974                        </div>
6975                    </div>
6976                    <br>
6977                    <div class="w3-content" style="height:10px"  id="Speech and Audio Modelling"></div>
6978                    <div class="w3-card w3-round w3-white w3-padding">
6979                        <div class="w3-container"  style="margin-top:40px">
6980                            <h4 class="w3-center">Speech and Audio Modelling</h4>
6981                            <hr>
6982                            <a class="w3-text" href="gao24_interspeech.html">
6983                                <p>
6984                                    GenDistiller: Distilling Pre-trained Language Models based on an Autoregressive Generative Model
6985                                    <br>
6986                                    <span class="w3-text w3-text-theme">
6987                                        Yingying Gao, Shilei Zhang, Chao Deng, Junlan Feng
6988                                    </span>
6989                                </p>
6990                            </a>
6991                            <a class="w3-text" href="guillaume24_interspeech.html">
6992                                <p>
6993                                    Gender and Language Identification in Multilingual Models of Speech: Exploring the Genericity and Robustness of Speech Representations
6994                                    <br>
6995                                    <span class="w3-text w3-text-theme">
6996                                        Séverine Guillaume, Maxime Fily, Alexis Michaud, Guillaume Wisniewski
6997                                    </span>
6998                                </p>
6999                            </a>
7000                            <a class="w3-text" href="wang24u_interspeech.html">
7001                                <p>
7002                                    Neural Compression Augmentation for Contrastive Audio Representation Learning
7003                                    <br>
7004                                    <span class="w3-text w3-text-theme">
7005                                        Zhaoyu Wang, Haohe Liu, Harry Coppock, Björn Schuller, Mark D. Plumbley
7006                                    </span>
7007                                </p>
7008                            </a>
7009                            <a class="w3-text" href="aluru24_interspeech.html">
7010                                <p>
7011                                    Post-Net: A linguistically inspired sequence-dependent transformed neural architecture for automatic syllable stress detection
7012                                    <br>
7013                                    <span class="w3-text w3-text-theme">
7014                                        Sai Harshitha Aluru, Jhansi Mallela, Chiranjeevi Yarra
7015                                    </span>
7016                                </p>
7017                            </a>
7018                        </div>
7019                    </div>
7020                    <br>
7021                    <div class="w3-content" style="height:10px"  id="Multi-Channel Speech Enhancement"></div>
7022                    <div class="w3-card w3-round w3-white w3-padding">
7023                        <div class="w3-container"  style="margin-top:40px">
7024                            <h4 class="w3-center">Multi-Channel Speech Enhancement</h4>
7025                            <hr>
7026                            <a class="w3-text" href="tammen24_interspeech.html">
7027                                <p>
7028                                    Array Geometry-Robust Attention-Based Neural Beamformer for Moving Speakers
7029                                    <br>
7030                                    <span class="w3-text w3-text-theme">
7031                                        Marvin Tammen, Tsubasa Ochiai, Marc Delcroix, Tomohiro Nakatani, Shoko Araki, Simon Doclo
7032                                    </span>
7033                                </p>
7034                            </a>
7035                            <a class="w3-text" href="xu24i_interspeech.html">
7036                                <p>
7037                                    FoVNet: Configurable Field-of-View Speech Enhancement with Low Computation and Distortion for Smart Glasses
7038                                    <br>
7039                                    <span class="w3-text w3-text-theme">
7040                                        Zhongweiyang Xu, Ali Aroudi, Ke Tan, Ashutosh Pandey, Jung-Suk Lee, Buye Xu, Francesco Nesta
7041                                    </span>
7042                                </p>
7043                            </a>
7044                            <a class="w3-text" href="aziz24_interspeech.html">
7045                                <p>
7046                                    Audio Enhancement from Multiple Crowdsourced Recordings: A Simple and Effective Baseline
7047                                    <br>
7048                                    <span class="w3-text w3-text-theme">
7049                                        Shiran Aziz, Yossi Adi, Shmuel Peleg
7050                                    </span>
7051                                </p>
7052                            </a>
7053                            <a class="w3-text" href="lee24g_interspeech.html">
7054                                <p>
7055                                    DeFTAN-AA: Array Geometry Agnostic Multichannel Speech Enhancement
7056                                    <br>
7057                                    <span class="w3-text w3-text-theme">
7058                                        Dongheon Lee, Jung-Woo Choi
7059                                    </span>
7060                                </p>
7061                            </a>
7062                            <a class="w3-text" href="zhou24d_interspeech.html">
7063                                <p>
7064                                    PLDNet: PLD-Guided Lightweight Deep Network Boosted by Efficient Attention for Handheld Dual-Microphone Speech Enhancement
7065                                    <br>
7066                                    <span class="w3-text w3-text-theme">
7067                                        Nan Zhou, Youhai Jiang, Jialin Tan, Chongmin Qi
7068                                    </span>
7069                                </p>
7070                            </a>
7071                        </div>
7072                    </div>
7073                    <br>
7074                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Paradigms and Methods 1"></div>
7075                    <div class="w3-card w3-round w3-white w3-padding">
7076                        <div class="w3-container"  style="margin-top:40px">
7077                            <h4 class="w3-center">Speech Synthesis: Paradigms and Methods 1</h4>
7078                            <hr>
7079                            <a class="w3-text" href="mcghee24_interspeech.html">
7080                                <p>
7081                                    Highly Intelligible Speaker-Independent Articulatory Synthesis
7082                                    <br>
7083                                    <span class="w3-text w3-text-theme">
7084                                        Charles McGhee, Kate Knill, Mark Gales
7085                                    </span>
7086                                </p>
7087                            </a>
7088                            <a class="w3-text" href="murata24_interspeech.html">
7089                                <p>
7090                                    An Attribute Interpolation Method in Speech Synthesis by Model Merging
7091                                    <br>
7092                                    <span class="w3-text w3-text-theme">
7093                                        Masato Murata, Koichi Miyazaki, Tomoki Koriyama
7094                                    </span>
7095                                </p>
7096                            </a>
7097                            <a class="w3-text" href="nishihara24_interspeech.html">
7098                                <p>
7099                                    Low-dimensional Style Token Control for Hyperarticulated Speech Synthesis
7100                                    <br>
7101                                    <span class="w3-text w3-text-theme">
7102                                        Miku Nishihara, Dan Wells, Korin Richmond, Aidan Pine
7103                                    </span>
7104                                </p>
7105                            </a>
7106                            <a class="w3-text" href="li24ba_interspeech.html">
7107                                <p>
7108                                    Single-Codec: Single-Codebook Speech Codec towards High-Performance Speech Generation
7109                                    <br>
7110                                    <span class="w3-text w3-text-theme">
7111                                        Hanzhao Li, Liumeng Xue, Haohan Guo, Xinfa Zhu, Yuanjun Lv, Lei Xie, Yunlin Chen, Hao Yin, Zhifei Li
7112                                    </span>
7113                                </p>
7114                            </a>
7115                            <a class="w3-text" href="dang24_interspeech.html">
7116                                <p>
7117                                    LiveSpeech: Low-Latency Zero-shot Text-to-Speech via Autoregressive Modeling of Audio Discrete Codes
7118                                    <br>
7119                                    <span class="w3-text w3-text-theme">
7120                                        Trung Dang, David Aponte, Dung Tran, Kazuhito Koishida
7121                                    </span>
7122                                </p>
7123                            </a>
7124                            <a class="w3-text" href="kim24h_interspeech.html">
7125                                <p>
7126                                    ClariTTS: Feature-ratio Normalization and Duration Stabilization for Code-mixed Multi-speaker Speech Synthesis
7127                                    <br>
7128                                    <span class="w3-text w3-text-theme">
7129                                        Changhwan Kim
7130                                    </span>
7131                                </p>
7132                            </a>
7133                            <a class="w3-text" href="janiczek24_interspeech.html">
7134                                <p>
7135                                    Multi-modal Adversarial Training for Zero-Shot Voice Cloning
7136                                    <br>
7137                                    <span class="w3-text w3-text-theme">
7138                                        John Janiczek, Dading Chong, Dongyang Dai, Arlo Faria, Chao Wang, Tao Wang, Yuzong Liu
7139                                    </span>
7140                                </p>
7141                            </a>
7142                            <a class="w3-text" href="chien24b_interspeech.html">
7143                                <p>
7144                                    Learning Fine-Grained Controllability on Speech Generation via Efficient Fine-Tuning
7145                                    <br>
7146                                    <span class="w3-text w3-text-theme">
7147                                        Chung-Ming Chien, Andros Tjandra, Apoorv Vyas, Matt Le, Bowen Shi, Wei-Ning Hsu
7148                                    </span>
7149                                </p>
7150                            </a>
7151                            <a class="w3-text" href="wu24o_interspeech.html">
7152                                <p>
7153                                    Modeling Vocal Tract Like Acoustic Tubes Using the Immersed Boundary Method
7154                                    <br>
7155                                    <span class="w3-text w3-text-theme">
7156                                        Rongshuai Wu, Debasish Ray Mohapatra, Sidney Fels
7157                                    </span>
7158                                </p>
7159                            </a>
7160                        </div>
7161                    </div>
7162                    <br>
7163                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Paradigms and Methods 2"></div>
7164                    <div class="w3-card w3-round w3-white w3-padding">
7165                        <div class="w3-container"  style="margin-top:40px">
7166                            <h4 class="w3-center">Speech Synthesis: Paradigms and Methods 2</h4>
7167                            <hr>
7168                            <a class="w3-text" href="lemerle24_interspeech.html">
7169                                <p>
7170                                    Small-E: Small Language Model with Linear Attention for Efficient Speech Synthesis
7171                                    <br>
7172                                    <span class="w3-text w3-text-theme">
7173                                        Théodor Lemerle, Nicolas Obin, Axel Roebel
7174                                    </span>
7175                                </p>
7176                            </a>
7177                            <a class="w3-text" href="neekhara24_interspeech.html">
7178                                <p>
7179                                    Improving Robustness of LLM-based Speech Synthesis by Learning Monotonic Alignment
7180                                    <br>
7181                                    <span class="w3-text w3-text-theme">
7182                                        Paarth Neekhara, Shehzeen Hussain, Subhankar Ghosh, Jason Li, Boris Ginsburg
7183                                    </span>
7184                                </p>
7185                            </a>
7186                            <a class="w3-text" href="lai24b_interspeech.html">
7187                                <p>
7188                                    Synthesizing Long-Form Speech merely from Sentence-Level Corpus with Content Extrapolation and LLM Contextual Enrichment
7189                                    <br>
7190                                    <span class="w3-text w3-text-theme">
7191                                        Shijie Lai, Minglu He, Zijing Zhao, Kai Wang, Hao Huang, Jichen Yang
7192                                    </span>
7193                                </p>
7194                            </a>
7195                            <a class="w3-text" href="liu24p_interspeech.html">
7196                                <p>
7197                                    FluentEditor: Text-based Speech Editing by Considering Acoustic and Prosody Consistency
7198                                    <br>
7199                                    <span class="w3-text w3-text-theme">
7200                                        Rui Liu, Jiatian Xi, Ziyue Jiang, Haizhou Li
7201                                    </span>
7202                                </p>
7203                            </a>
7204                            <a class="w3-text" href="zhou24_interspeech.html">
7205                                <p>
7206                                    Phonetic Enhanced Language Modeling for Text-to-Speech Synthesis
7207                                    <br>
7208                                    <span class="w3-text w3-text-theme">
7209                                        Kun Zhou, Shengkui Zhao, Yukun Ma, Chong Zhang, Hao Wang, Dianwen Ng, Chongjia Ni, Trung Hieu Nguyen, Jia Qi Yip, Bin Ma
7210                                    </span>
7211                                </p>
7212                            </a>
7213                            <a class="w3-text" href="lee24f_interspeech.html">
7214                                <p>
7215                                    High Fidelity Text-to-Speech Via Discrete Tokens Using Token Transducer and Group Masked Language Model
7216                                    <br>
7217                                    <span class="w3-text w3-text-theme">
7218                                        Joun Yeop Lee, Myeonghun Jeong, Minchan Kim, Ji-Hyun Lee, Hoon-Young Cho, Nam Soo Kim
7219                                    </span>
7220                                </p>
7221                            </a>
7222                            <a class="w3-text" href="lenglet24_interspeech.html">
7223                                <p>
7224                                    FastLips: an End-to-End Audiovisual Text-to-Speech System with Lip Features Prediction for Virtual Avatars
7225                                    <br>
7226                                    <span class="w3-text w3-text-theme">
7227                                        Martin Lenglet, Olivier Perrotin, Gerard Bailly
7228                                    </span>
7229                                </p>
7230                            </a>
7231                        </div>
7232                    </div>
7233                    <br>
7234                    <div class="w3-content" style="height:10px"  id="Neural Network Architectures for ASR 1"></div>
7235                    <div class="w3-card w3-round w3-white w3-padding">
7236                        <div class="w3-container"  style="margin-top:40px">
7237                            <h4 class="w3-center">Neural Network Architectures for ASR 1</h4>
7238                            <hr>
7239                            <a class="w3-text" href="yang24g_interspeech.html">
7240                                <p>
7241                                    Contemplative Mechanism for Speech Recognition: Speech Encoders can Think
7242                                    <br>
7243                                    <span class="w3-text w3-text-theme">
7244                                        Tien-Ju Yang, Andrew Rosenberg, Bhuvana Ramabhadran
7245                                    </span>
7246                                </p>
7247                            </a>
7248                            <a class="w3-text" href="parcollet24_interspeech.html">
7249                                <p>
7250                                    SummaryMixing: A Linear-Complexity Alternative to Self-Attention for Speech Recognition and Understanding
7251                                    <br>
7252                                    <span class="w3-text w3-text-theme">
7253                                        Titouan Parcollet, Rogier van Dalen, Shucong Zhang, Sourav Bhattacharya
7254                                    </span>
7255                                </p>
7256                            </a>
7257                            <a class="w3-text" href="moriya24_interspeech.html">
7258                                <p>
7259                                    Boosting Hybrid Autoregressive Transducer-based ASR with Internal Acoustic Model Training and Dual Blank Thresholding
7260                                    <br>
7261                                    <span class="w3-text w3-text-theme">
7262                                        Takafumi Moriya, Takanori Ashihara, Masato Mimura, Hiroshi Sato, Kohei Matsuura, Ryo Masumura, Taichi Asami
7263                                    </span>
7264                                </p>
7265                            </a>
7266                            <a class="w3-text" href="kundu24_interspeech.html">
7267                                <p>
7268                                    RepCNN: Micro-sized, Mighty Models for Wakeword Detection
7269                                    <br>
7270                                    <span class="w3-text w3-text-theme">
7271                                        Arnav Kundu, Prateeth Nayak, Priyanka Padmanabhan, Devang Naik
7272                                    </span>
7273                                </p>
7274                            </a>
7275                            <a class="w3-text" href="vankeirsbilck24_interspeech.html">
7276                                <p>
7277                                    Conformer without Convolutions
7278                                    <br>
7279                                    <span class="w3-text w3-text-theme">
7280                                        Matthijs Van keirsbilck, Alexander Keller
7281                                    </span>
7282                                </p>
7283                            </a>
7284                            <a class="w3-text" href="zhang24e_interspeech.html">
7285                                <p>
7286                                    Linear-Complexity Self-Supervised Learning for Speech Processing
7287                                    <br>
7288                                    <span class="w3-text w3-text-theme">
7289                                        Shucong Zhang, Titouan Parcollet, Rogier van Dalen, Sourav Bhattacharya
7290                                    </span>
7291                                </p>
7292                            </a>
7293                        </div>
7294                    </div>
7295                    <br>
7296                    <div class="w3-content" style="height:10px"  id="Error Correction and Rescoring"></div>
7297                    <div class="w3-card w3-round w3-white w3-padding">
7298                        <div class="w3-container"  style="margin-top:40px">
7299                            <h4 class="w3-center">Error Correction and Rescoring</h4>
7300                            <hr>
7301                            <a class="w3-text" href="mittal24_interspeech.html">
7302                                <p>
7303                                    SALSA: Speedy ASR-LLM Synchronous Aggregation
7304                                    <br>
7305                                    <span class="w3-text w3-text-theme">
7306                                        Ashish Mittal, Darshan Prabhu, Sunita Sarawagi, Preethi Jyothi
7307                                    </span>
7308                                </p>
7309                            </a>
7310                            <a class="w3-text" href="yoon24c_interspeech.html">
7311                                <p>
7312                                    LI-TTA: Language Informed Test-Time Adaptation for Automatic Speech Recognition
7313                                    <br>
7314                                    <span class="w3-text w3-text-theme">
7315                                        Eunseop Yoon, Hee Suk Yoon, John Harvill, Mark Hasegawa-Johnson, Chang D. Yoo
7316                                    </span>
7317                                </p>
7318                            </a>
7319                            <a class="w3-text" href="wang24j_interspeech.html">
7320                                <p>
7321                                    HypR: A comprehensive study for ASR hypothesis revising with a reference corpus
7322                                    <br>
7323                                    <span class="w3-text w3-text-theme">
7324                                        Yi-Wei Wang, Ke-Han Lu, Kuan-Yu Chen
7325                                    </span>
7326                                </p>
7327                            </a>
7328                            <a class="w3-text" href="shu24_interspeech.html">
7329                                <p>
7330                                    Error Correction by Paying Attention to Both Acoustic and Confidence References for Automatic Speech Recognition
7331                                    <br>
7332                                    <span class="w3-text w3-text-theme">
7333                                        Yuchun Shu, Bo Hu, Yifeng He, Hao Shi, Longbiao Wang, Jianwu Dang
7334                                    </span>
7335                                </p>
7336                            </a>
7337                            <a class="w3-text" href="kang24c_interspeech.html">
7338                                <p>
7339                                    Transformer-based Model for ASR N-Best Rescoring and Rewriting
7340                                    <br>
7341                                    <span class="w3-text w3-text-theme">
7342                                        Iwen E Kang, Christophe Van Gysel, Man-Hung Siu
7343                                    </span>
7344                                </p>
7345                            </a>
7346                            <a class="w3-text" href="yang24b_interspeech.html">
7347                                <p>
7348                                    RASU: Retrieval Augmented Speech Understanding through Generative Modeling
7349                                    <br>
7350                                    <span class="w3-text w3-text-theme">
7351                                        Hao Yang, Min Zhang, Minghan Wang, Jiaxin Guo
7352                                    </span>
7353                                </p>
7354                            </a>
7355                        </div>
7356                    </div>
7357                    <br>
7358                    <div class="w3-content" style="height:10px"  id="Spoken Language Understanding"></div>
7359                    <div class="w3-card w3-round w3-white w3-padding">
7360                        <div class="w3-container"  style="margin-top:40px">
7361                            <h4 class="w3-center">Spoken Language Understanding</h4>
7362                            <hr>
7363                            <a class="w3-text" href="aimaiti24_interspeech.html">
7364                                <p>
7365                                    An Uyghur Extension to the MASSIVE Multi-lingual Spoken Language Understanding Corpus with Comprehensive Evaluations
7366                                    <br>
7367                                    <span class="w3-text w3-text-theme">
7368                                        Ainikaerjiang Aimaiti, Di Wu, Liting Jiang, Gulinigeer Abudouwaili, Hao Huang, Wushour Silamu
7369                                    </span>
7370                                </p>
7371                            </a>
7372                            <a class="w3-text" href="christ24_interspeech.html">
7373                                <p>
7374                                    This Paper Had the Smartest Reviewers - Flattery Detection Utilising an Audio-Textual Transformer-Based Approach
7375                                    <br>
7376                                    <span class="w3-text w3-text-theme">
7377                                        Lukas Christ, Shahin Amiriparian, Friederike Hawighorst, Ann-Kathrin Schill, Angelo Boutalikakis, Lorenz Graf-Vlachy, Andreas König, Björn Schuller
7378                                    </span>
7379                                </p>
7380                            </a>
7381                            <a class="w3-text" href="akani24_interspeech.html">
7382                                <p>
7383                                    Unified Framework for Spoken Language Understanding and Summarization in Task-Based Human Dialog processing
7384                                    <br>
7385                                    <span class="w3-text w3-text-theme">
7386                                        Eunice Akani, Frederic Bechet, Benoît Favre, Romain Gemignani
7387                                    </span>
7388                                </p>
7389                            </a>
7390                            <a class="w3-text" href="anderson24_interspeech.html">
7391                                <p>
7392                                    Automated Human-Readable Label Generation in Open Intent Discovery
7393                                    <br>
7394                                    <span class="w3-text w3-text-theme">
7395                                        Grant Anderson, Emma Hart, Dimitra Gkatzia, Ian Beaver
7396                                    </span>
7397                                </p>
7398                            </a>
7399                            <a class="w3-text" href="chang24_interspeech.html">
7400                                <p>
7401                                    Applying Reinforcement Learning and Multi-Generators for Stage Transition in an Emotional Support Dialogue System
7402                                    <br>
7403                                    <span class="w3-text w3-text-theme">
7404                                        Jeremy Chang, Kuan-Yu Chen, Chung-Hsien Wu
7405                                    </span>
7406                                </p>
7407                            </a>
7408                        </div>
7409                    </div>
7410                    <br>
7411                    <div class="w3-content" style="height:10px"  id="Spoken Dialogue Systems and Conversational Analysis 2"></div>
7412                    <div class="w3-card w3-round w3-white w3-padding">
7413                        <div class="w3-container"  style="margin-top:40px">
7414                            <h4 class="w3-center">Spoken Dialogue Systems and Conversational Analysis 2</h4>
7415                            <hr>
7416                            <a class="w3-text" href="chen24d_interspeech.html">
7417                                <p>
7418                                    Target conversation extraction: Source separation using turn-taking dynamics
7419                                    <br>
7420                                    <span class="w3-text w3-text-theme">
7421                                        Tuochao Chen, Qirui Wang, Bohan Wu, Malek Itani, Emre Sefik Eskimez, Takuya Yoshioka, Shyamnath Gollakota
7422                                    </span>
7423                                </p>
7424                            </a>
7425                            <a class="w3-text" href="ng24b_interspeech.html">
7426                                <p>
7427                                    Investigating the Influence of Stance-Taking on Conversational Timing of Task-Oriented Speech
7428                                    <br>
7429                                    <span class="w3-text w3-text-theme">
7430                                        Sara Ng, Gina-Anne Levow, Mari Ostendorf, Richard Wright
7431                                    </span>
7432                                </p>
7433                            </a>
7434                            <a class="w3-text" href="uro24_interspeech.html">
7435                                <p>
7436                                    Detecting the terminality of speech-turn boundary for spoken interactions in French TV and Radio content
7437                                    <br>
7438                                    <span class="w3-text w3-text-theme">
7439                                        Rémi Uro, Marie Tahon, David Doukhan, Antoine Laurent, Albert Rilliard
7440                                    </span>
7441                                </p>
7442                            </a>
7443                            <a class="w3-text" href="watanabe24_interspeech.html">
7444                                <p>
7445                                    Utilization of Text Data for Response Timing Detection in Attentive Listening
7446                                    <br>
7447                                    <span class="w3-text w3-text-theme">
7448                                        Yu Watanabe, Koichiro Ito, Shigeki Matsubara
7449                                    </span>
7450                                </p>
7451                            </a>
7452                            <a class="w3-text" href="park24b_interspeech.html">
7453                                <p>
7454                                    Backchannel prediction, based on who, when and what
7455                                    <br>
7456                                    <span class="w3-text w3-text-theme">
7457                                        Yo-Han Park, Wencke Liermann, Yong-Seok Choi, Seung Hi Kim, Jeong-Uk Bang, Seung Yun, Kong Joo Lee
7458                                    </span>
7459                                </p>
7460                            </a>
7461                            <a class="w3-text" href="hutin24_interspeech.html">
7462                                <p>
7463                                    Uh, um and mh: Are filled pauses prone to conversational converge?
7464                                    <br>
7465                                    <span class="w3-text w3-text-theme">
7466                                        Mathilde Hutin, Junfei Hu, Liesbeth Degand
7467                                    </span>
7468                                </p>
7469                            </a>
7470                            <a class="w3-text" href="ohagi24_interspeech.html">
7471                                <p>
7472                                    Investigation of look-ahead techniques to improve response time in spoken dialogue system
7473                                    <br>
7474                                    <span class="w3-text w3-text-theme">
7475                                        Masaya Ohagi, Tomoya Mizumoto, Katsumasa Yoshikawa
7476                                    </span>
7477                                </p>
7478                            </a>
7479                        </div>
7480                    </div>
7481                    <br>
7482                    <div class="w3-content" style="height:10px"  id="Computational Models of Human Language Acquisition, Perception, and Production (Special Session)"></div>
7483                    <div class="w3-card w3-round w3-white w3-padding">
7484                        <div class="w3-container"  style="margin-top:40px">
7485                            <h4 class="w3-center">Computational Models of Human Language Acquisition, Perception, and Production (Special Session)</h4>
7486                            <hr>
7487                            <a class="w3-text" href="heuser24b_interspeech.html">
7488                                <p>
7489                                    Information-theoretic hypothesis generation of relative cue weighting for the voicing contrast
7490                                    <br>
7491                                    <span class="w3-text w3-text-theme">
7492                                        Annika Heuser, Jianjing Kuang
7493                                    </span>
7494                                </p>
7495                            </a>
7496                            <a class="w3-text" href="hovsepyan24_interspeech.html">
7497                                <p>
7498                                    Neurocomputational model of speech recognition for pathological speech detection: a case study on Parkinson's disease speech detection
7499                                    <br>
7500                                    <span class="w3-text w3-text-theme">
7501                                        Sevada Hovsepyan, Mathew Magimai.-Doss
7502                                    </span>
7503                                </p>
7504                            </a>
7505                            <a class="w3-text" href="ortiztandazo24_interspeech.html">
7506                                <p>
7507                                    Simulating articulatory trajectories with phonological feature interpolation
7508                                    <br>
7509                                    <span class="w3-text w3-text-theme">
7510                                        Angelo Ortiz Tandazo, Thomas Schatz, Thomas Hueber, Emmanuel Dupoux
7511                                    </span>
7512                                </p>
7513                            </a>
7514                            <a class="w3-text" href="onda24_interspeech.html">
7515                                <p>
7516                                    A Pilot Study of GSLM-based Simulation of Foreign Accentuation Only Using Native Speech Corpora
7517                                    <br>
7518                                    <span class="w3-text w3-text-theme">
7519                                        Kentaro Onda, Joonyong Park, Nobuaki Minematsu, Daisuke Saito
7520                                    </span>
7521                                </p>
7522                            </a>
7523                            <a class="w3-text" href="bonafos24_interspeech.html">
7524                                <p>
7525                                    Dirichlet process mixture model based on topologically augmented signal representation for clustering infant vocalizations
7526                                    <br>
7527                                    <span class="w3-text w3-text-theme">
7528                                        Guillem Bonafos, Clara Bourot, Pierre Pudlo, Jean-Marc Freyermuth, Laurence Reboul, Samuel Tronçon, Arnaud Rey
7529                                    </span>
7530                                </p>
7531                            </a>
7532                            <a class="w3-text" href="elie24_interspeech.html">
7533                                <p>
7534                                    A data-driven model of acoustic speech intelligibility for optimization-based models of speech production
7535                                    <br>
7536                                    <span class="w3-text w3-text-theme">
7537                                        Benjamin Elie, Juraj Simko, Alice Turk
7538                                    </span>
7539                                </p>
7540                            </a>
7541                            <a class="w3-text" href="coffey24_interspeech.html">
7542                                <p>
7543                                    The Difficulty and Importance of Estimating the Lower and Upper Bounds of Infant Speech Exposure
7544                                    <br>
7545                                    <span class="w3-text w3-text-theme">
7546                                        Joseph Coffey, Okko Räsänen, Camila Scaff, Alejandrina Cristia
7547                                    </span>
7548                                </p>
7549                            </a>
7550                            <a class="w3-text" href="vanniekerk24_interspeech.html">
7551                                <p>
7552                                    Spoken-Term Discovery using Discrete Speech Units
7553                                    <br>
7554                                    <span class="w3-text w3-text-theme">
7555                                        Benjamin van Niekerk, Julian Zaïdi, Marc-André Carbonneau, Herman Kamper
7556                                    </span>
7557                                </p>
7558                            </a>
7559                            <a class="w3-text" href="mohamed24_interspeech.html">
7560                                <p>
7561                                    Orthogonality and isotropy of speaker and phonetic information in self-supervised speech representations
7562                                    <br>
7563                                    <span class="w3-text w3-text-theme">
7564                                        Mukhtar Mohamed, Oli Danyi Liu, Hao Tang, Sharon Goldwater
7565                                    </span>
7566                                </p>
7567                            </a>
7568                        </div>
7569                    </div>
7570                    <br>
7571                    <div class="w3-content" style="height:10px"  id="Show and Tell 3"></div>
7572                    <div class="w3-card w3-round w3-white w3-padding">
7573                        <div class="w3-container"  style="margin-top:40px">
7574                            <h4 class="w3-center">Show and Tell 3</h4>
7575                            <hr>
7576                            <a class="w3-text" href="ryumina24_interspeech.html">
7577                                <p>
7578                                    OCEAN-AI: open multimodal framework for personality traits assessment and HR-processes automatization
7579                                    <br>
7580                                    <span class="w3-text w3-text-theme">
7581                                        Elena Ryumina, Dmitry Ryumin, Alexey Karpov
7582                                    </span>
7583                                </p>
7584                            </a>
7585                            <a class="w3-text" href="mundra24_interspeech.html">
7586                                <p>
7587                                    VoxMed: one-step respiratory disease classifier using digital stethoscope sounds
7588                                    <br>
7589                                    <span class="w3-text w3-text-theme">
7590                                        Paridhi Mundra, Manik Sharma, Yashwardhan Chaudhuri, Orchid Chetia Phukan, Arun Balaji Buduru
7591                                    </span>
7592                                </p>
7593                            </a>
7594                            <a class="w3-text" href="sharma24b_interspeech.html">
7595                                <p>
7596                                    AVR: synergizing foundation models for audio-visual humor detection
7597                                    <br>
7598                                    <span class="w3-text w3-text-theme">
7599                                        Sarthak Sharma, Orchid Chetia Phukan, Drishti Singh, Arun Balaji Buduru, Rajesh Sharma
7600                                    </span>
7601                                </p>
7602                            </a>
7603                            <a class="w3-text" href="chaudhuri24_interspeech.html">
7604                                <p>
7605                                    ASGIR: audio spectrogram transformer guided classification and information retrieval for birds
7606                                    <br>
7607                                    <span class="w3-text w3-text-theme">
7608                                        Yashwardhan Chaudhuri, Paridhi Mundra, Arnesh Batra, Orchid Chetia Phukan, Arun Balaji Buduru
7609                                    </span>
7610                                </p>
7611                            </a>
7612                            <a class="w3-text" href="koshal24_interspeech.html">
7613                                <p>
7614                                    PERSONA: an application for emotion recognition, gender recognition and age estimation
7615                                    <br>
7616                                    <span class="w3-text w3-text-theme">
7617                                        Devyani Koshal, Orchid Chetia Phukan, Sarthak Jain, Arun Balaji Buduru, Rajesh Sharma
7618                                    </span>
7619                                </p>
7620                            </a>
7621                            <a class="w3-text" href="akhtar24_interspeech.html">
7622                                <p>
7623                                    NeuRO: an application for code-switched autism detection in children
7624                                    <br>
7625                                    <span class="w3-text w3-text-theme">
7626                                        Mohd Mujtaba Akhtar,  Girish, Orchid Chetia Phukan, Muskaan Singh
7627                                    </span>
7628                                </p>
7629                            </a>
7630                            <a class="w3-text" href="phukan24c_interspeech.html">
7631                                <p>
7632                                    ComFeAT: combination of neural and spectral features for improved depression detection
7633                                    <br>
7634                                    <span class="w3-text w3-text-theme">
7635                                        Orchid Chetia Phukan, Sarthak Jain, Shubham Singh, Muskaan Singh, Arun Balaji Buduru, Rajesh Sharma
7636                                    </span>
7637                                </p>
7638                            </a>
7639                            <a class="w3-text" href="jain24b_interspeech.html">
7640                                <p>
7641                                    The reasonable effectiveness of speaker embeddings for violence detection
7642                                    <br>
7643                                    <span class="w3-text w3-text-theme">
7644                                        Sarthak Jain, Orchid Chetia Phukan, Arun Balaji Buduru, Rajesh Sharma
7645                                    </span>
7646                                </p>
7647                            </a>
7648                            <a class="w3-text" href="obukhov24_interspeech.html">
7649                                <p>
7650                                    ATTEST: an analytics tool for the testing and evaluation of speech technologies
7651                                    <br>
7652                                    <span class="w3-text w3-text-theme">
7653                                        Dmitrii Obukhov, Marcel de Korte, Andrey Adaschik
7654                                    </span>
7655                                </p>
7656                            </a>
7657                            <a class="w3-text" href="masson24_interspeech.html">
7658                                <p>
7659                                    PhoneViz: exploring alignments at a glance
7660                                    <br>
7661                                    <span class="w3-text w3-text-theme">
7662                                        Margot Masson, Erfan A. Shams, Iona Gessinger, Julie Carson-Berndsen
7663                                    </span>
7664                                </p>
7665                            </a>
7666                            <a class="w3-text" href="pages24_interspeech.html">
7667                                <p>
7668                                    Gryannote open-source speaker diarization labeling tool
7669                                    <br>
7670                                    <span class="w3-text w3-text-theme">
7671                                        Clément Pages, Hervé Bredin
7672                                    </span>
7673                                </p>
7674                            </a>
7675                            <a class="w3-text" href="morrone24_interspeech.html">
7676                                <p>
7677                                    A toolkit for joint speaker diarization and identification with application to speaker-attributed ASR
7678                                    <br>
7679                                    <span class="w3-text w3-text-theme">
7680                                        Giovanni Morrone, Enrico Zovato, Fabio Brugnara, Enrico Sartori, Leonardo Badino
7681                                    </span>
7682                                </p>
7683                            </a>
7684                        </div>
7685                    </div>
7686                    <br>
7687                    <div class="w3-content" style="height:10px"  id="Phonetics, Phonology and Prosody"></div>
7688                    <div class="w3-card w3-round w3-white w3-padding">
7689                        <div class="w3-container"  style="margin-top:40px">
7690                            <h4 class="w3-center">Phonetics, Phonology and Prosody</h4>
7691                            <hr>
7692                            <a class="w3-text" href="kinnunen24_interspeech.html">
7693                                <p>
7694                                    Speaker Detection by the Individual Listener and the Crowd: Parametric Models Applicable to Bonafide and Deepfake Speech
7695                                    <br>
7696                                    <span class="w3-text w3-text-theme">
7697                                        Tomi H. Kinnunen, Rosa Gonzalez Hautamäki, Xin Wang, Junichi Yamagishi
7698                                    </span>
7699                                </p>
7700                            </a>
7701                            <a class="w3-text" href="deluca24_interspeech.html">
7702                                <p>
7703                                    NumberLie: a game-based experiment to understand the acoustics of deception and truthfulness
7704                                    <br>
7705                                    <span class="w3-text w3-text-theme">
7706                                        Alessandro De Luca, Andrew Clark, Volker Dellwo
7707                                    </span>
7708                                </p>
7709                            </a>
7710                            <a class="w3-text" href="loiacono24_interspeech.html">
7711                                <p>
7712                                    Preservation, conservation and phonetic study of the voices of Italian poets: A study on the seven years of the VIP archive
7713                                    <br>
7714                                    <span class="w3-text w3-text-theme">
7715                                        Federico Lo Iacono, Valentina Colonna, Antonio Romano
7716                                    </span>
7717                                </p>
7718                            </a>
7719                            <a class="w3-text" href="audibert24_interspeech.html">
7720                                <p>
7721                                    Do Speaker-dependent Vowel Characteristics depend on Speech Style?
7722                                    <br>
7723                                    <span class="w3-text w3-text-theme">
7724                                        Nicolas Audibert, Cecile Fougeron, Christine Meunier
7725                                    </span>
7726                                </p>
7727                            </a>
7728                            <a class="w3-text" href="liu24q_interspeech.html">
7729                                <p>
7730                                    A comparison of voice similarity through acoustics, human perception and deep neural network (DNN) speaker verification systems
7731                                    <br>
7732                                    <span class="w3-text w3-text-theme">
7733                                        Suyuan Liu, Molly Babel, Jian Zhu
7734                                    </span>
7735                                </p>
7736                            </a>
7737                            <a class="w3-text" href="jones24_interspeech.html">
7738                                <p>
7739                                    Evaluating Italian Vowel Variation with the Recurrent Neural Network Phonet
7740                                    <br>
7741                                    <span class="w3-text w3-text-theme">
7742                                        Austin Jones, Margaret E. L. Renwick
7743                                    </span>
7744                                </p>
7745                            </a>
7746                            <a class="w3-text" href="tulchynska24_interspeech.html">
7747                                <p>
7748                                    Prosodic marking of syntactic boun
7748daries in Khoekhoe
7749                                    <br>
7750                                    <span class="w3-text w3-text-theme">
7751                                        Kira Tulchynska, Sylvanus Job, Alena Witzlack-Makarevich, Margaret Zellers
7752                                    </span>
7753                                </p>
7754                            </a>
7755                        </div>
7756                    </div>
7757                    <br>
7758                    <div class="w3-content" style="height:10px"  id="Segmentals"></div>
7759                    <div class="w3-card w3-round w3-white w3-padding">
7760                        <div class="w3-container"  style="margin-top:40px">
7761                            <h4 class="w3-center">Segmentals</h4>
7762                            <hr>
7763                            <a class="w3-text" href="cronenberg24_interspeech.html">
7764                                <p>
7765                                    Crosslinguistic Comparison of Acoustic Variation in the Vowel Sequences /ia/ and /io/ in Four Romance Languages
7766                                    <br>
7767                                    <span class="w3-text w3-text-theme">
7768                                        Johanna Cronenberg, Ioana Chitoran, Lori Lamel, Ioana Vasilescu
7769                                    </span>
7770                                </p>
7771                            </a>
7772                            <a class="w3-text" href="vegarodriguez24_interspeech.html">
7773                                <p>
7774                                    Nasal Air Flow During Speech Production In Korebaju
7775                                    <br>
7776                                    <span class="w3-text w3-text-theme">
7777                                        Jenifer Vega Rodriguez, Nathalie Vallée, Christophe Savariaux, Silvain Gerber
7778                                    </span>
7779                                </p>
7780                            </a>
7781                            <a class="w3-text" href="kye24_interspeech.html">
7782                                <p>
7783                                    Affricates in Lushootseed
7784                                    <br>
7785                                    <span class="w3-text w3-text-theme">
7786                                        Ted Kye
7787                                    </span>
7788                                </p>
7789                            </a>
7790                            <a class="w3-text" href="terhiija24_interspeech.html">
7791                                <p>
7792                                    Voiced and voiceless laterals in Angami
7793                                    <br>
7794                                    <span class="w3-text w3-text-theme">
7795                                        Viyazonuo Terhiija, Priyankoo Sarmah
7796                                    </span>
7797                                </p>
7798                            </a>
7799                            <a class="w3-text" href="yang24n_interspeech.html">
7800                                <p>
7801                                    Intrusive schwa within French stop-liquid clusters: An acoustic analysis
7802                                    <br>
7803                                    <span class="w3-text w3-text-theme">
7804                                        Minmin Yang, Rachid Ridouane
7805                                    </span>
7806                                </p>
7807                            </a>
7808                        </div>
7809                    </div>
7810                    <br>
7811                    <div class="w3-content" style="height:10px"  id="New Avenues in Emotion Recognition"></div>
7812                    <div class="w3-card w3-round w3-white w3-padding">
7813                        <div class="w3-container"  style="margin-top:40px">
7814                            <h4 class="w3-center">New Avenues in Emotion Recognition</h4>
7815                            <hr>
7816                            <a class="w3-text" href="wu24d_interspeech.html">
7817                                <p>
7818                                    Can Modelling Inter-Rater Ambiguity Lead To Noise-Robust Continuous Emotion Predictions?
7819                                    <br>
7820                                    <span class="w3-text w3-text-theme">
7821                                        Ya-Tse Wu, Jingyao Wu, Vidhyasaharan Sethu, Chi-Chun Lee
7822                                    </span>
7823                                </p>
7824                            </a>
7825                            <a class="w3-text" href="zhao24g_interspeech.html">
7826                                <p>
7827                                    MFDR: Multiple-stage Fusion and Dynamically Refined Network for Multimodal Emotion Recognition
7828                                    <br>
7829                                    <span class="w3-text w3-text-theme">
7830                                        Ziping Zhao, Tian Gao, Haishuai Wang, Björn Schuller
7831                                    </span>
7832                                </p>
7833                            </a>
7834                            <a class="w3-text" href="shi24i_interspeech.html">
7835                                <p>
7836                                    Multimodal Fusion of Music Theory-Inspired and Self-Supervised Representations for Improved Emotion Recognition
7837                                    <br>
7838                                    <span class="w3-text w3-text-theme">
7839                                        Xiaohan Shi, Xingfeng Li, Tomoki Toda
7840                                    </span>
7841                                </p>
7842                            </a>
7843                            <a class="w3-text" href="triantafyllopoulos24c_interspeech.html">
7844                                <p>
7845                                    Enrolment-based personalisation for improving individual-level fairness in speech emotion recognition
7846                                    <br>
7847                                    <span class="w3-text w3-text-theme">
7848                                        Andreas Triantafyllopoulos, Björn Schuller
7849                                    </span>
7850                                </p>
7851                            </a>
7852                            <a class="w3-text" href="leem24_interspeech.html">
7853                                <p>
7854                                    Keep, Delete, or Substitute: Frame Selection Strategy for Noise-Robust Speech Emotion Recognition
7855                                    <br>
7856                                    <span class="w3-text w3-text-theme">
7857                                        Seong-Gyun Leem, Daniel Fulford, Jukka-Pekka Onnela, David Gard, Carlos Busso
7858                                    </span>
7859                                </p>
7860                            </a>
7861                            <a class="w3-text" href="lu24e_interspeech.html">
7862                                <p>
7863                                    Hierarchical Distribution Adaptation for Unsupervised Cross-corpus Speech Emotion Recognition
7864                                    <br>
7865                                    <span class="w3-text w3-text-theme">
7866                                        Cheng Lu, Yuan Zong, Yan Zhao, Hailun Lian, Tianhua Qi, Björn Schuller, Wenming Zheng
7867                                    </span>
7868                                </p>
7869                            </a>
7870                        </div>
7871                    </div>
7872                    <br>
7873                    <div class="w3-content" style="height:10px"  id="Speaker Diarization 2"></div>
7874                    <div class="w3-card w3-round w3-white w3-padding">
7875                        <div class="w3-container"  style="margin-top:40px">
7876                            <h4 class="w3-center">Speaker Diarization 2</h4>
7877                            <hr>
7878                            <a class="w3-text" href="zhang24b_interspeech.html">
7879                                <p>
7880                                    Variable Segment Length and Domain-Adapted Feature Optimization for Speaker Diarization
7881                                    <br>
7882                                    <span class="w3-text w3-text-theme">
7883                                        Chenyuan Zhang, Linkai Luo, Hong Peng, Wei Wen
7884                                    </span>
7885                                </p>
7886                            </a>
7887                            <a class="w3-text" href="choi24d_interspeech.html">
7888                                <p>
7889                                    Efficient Speaker Embedding Extraction Using a Twofold Sliding Window Algorithm for Speaker Diarization
7890                                    <br>
7891                                    <span class="w3-text w3-text-theme">
7892                                        Jeong-Hwan Choi, Ye-Rin Jeoung, Ilseok Kim, Joon-Hyuk Chang
7893                                    </span>
7894                                </p>
7895                            </a>
7896                            <a class="w3-text" href="wang24h_interspeech.html">
7897                                <p>
7898                                    DiarizationLM: Speaker Diarization Post-Processing with Large Language Models
7899                                    <br>
7900                                    <span class="w3-text w3-text-theme">
7901                                        Quan Wang, Yiling Huang, Guanlong Zhao, Evan Clark, Wei Xia, Hank Liao
7902                                    </span>
7903                                </p>
7904                            </a>
7905                            <a class="w3-text" href="blatt24_interspeech.html">
7906                                <p>
7907                                    Joint vs Sequential Speaker-Role Detection and Automatic Speech Recognition for Air-traffic Control
7908                                    <br>
7909                                    <span class="w3-text w3-text-theme">
7910                                        Alexander Blatt, Aravind Krishnan, Dietrich Klakow
7911                                    </span>
7912                                </p>
7913                            </a>
7914                            <a class="w3-text" href="plaquet24_interspeech.html">
7915                                <p>
7916                                    On the calibration of powerset speaker diarization models
7917                                    <br>
7918                                    <span class="w3-text w3-text-theme">
7919                                        Alexis Plaquet, Hervé Bredin
7920                                    </span>
7921                                </p>
7922                            </a>
7923                            <a class="w3-text" href="baroudi24_interspeech.html">
7924                                <p>
7925                                    Specializing Self-Supervised Speech Representations for Speaker Segmentation
7926                                    <br>
7927                                    <span class="w3-text w3-text-theme">
7928                                        Séverin Baroudi, Thomas Pellegrini, Hervé Bredin
7929                                    </span>
7930                                </p>
7931                            </a>
7932                        </div>
7933                    </div>
7934                    <br>
7935                    <div class="w3-content" style="height:10px"  id="Speaker Recognition 2"></div>
7936                    <div class="w3-card w3-round w3-white w3-padding">
7937                        <div class="w3-container"  style="margin-top:40px">
7938                            <h4 class="w3-center">Speaker Recognition 2</h4>
7939                            <hr>
7940                            <a class="w3-text" href="loweimi24_interspeech.html">
7941                                <p>
7942                                    On the Usefulness of Speaker Embeddings for Speaker Retrieval in the Wild: A Comparative Study of x-vector and ECAPA-TDNN Models
7943                                    <br>
7944                                    <span class="w3-text w3-text-theme">
7945                                        Erfan Loweimi, Mengjie Qian, Kate Knill, Mark Gales
7946                                    </span>
7947                                </p>
7948                            </a>
7949                            <a class="w3-text" href="jin24b_interspeech.html">
7950                                <p>
7951                                    W-GVKT: Within-Global-View Knowledge Transfer for Speaker Verification
7952                                    <br>
7953                                    <span class="w3-text w3-text-theme">
7954                                        Zezhong Jin, Youzhi Tu, Man-Wai Mak
7955                                    </span>
7956                                </p>
7957                            </a>
7958                            <a class="w3-text" href="shen24_interspeech.html">
7959                                <p>
7960                                    CEC: A Noisy Label Detection Method for Speaker Recognition
7961                                    <br>
7962                                    <span class="w3-text w3-text-theme">
7963                                        Yao Shen, Yingying Gao, Yaqian Hao, Chenguang Hu, Fulin Zhang, Junlan Feng, Shilei Zhang
7964                                    </span>
7965                                </p>
7966                            </a>
7967                            <a class="w3-text" href="zhang24c_interspeech.html">
7968                                <p>
7969                                    Disentangling Age and Identity with a Mutual Information Minimization for Cross-Age Speaker Verification
7970                                    <br>
7971                                    <span class="w3-text w3-text-theme">
7972                                        Fengrun Zhang, Wangjin Zhou, Yiming Liu, Wang Geng, Yahui Shan, Chen Zhang
7973                                    </span>
7974                                </p>
7975                            </a>
7976                            <a class="w3-text" href="li24u_interspeech.html">
7977                                <p>
7978                                    Contrastive Learning and Inter-Speaker Distribution Alignment Based Unsupervised Domain Adaptation for Robust Speaker Verification
7979                                    <br>
7980                                    <span class="w3-text w3-text-theme">
7981                                        Zuoliang Li, Wu Guo, Bin Gu, Shengyu Peng, Jie Zhang
7982                                    </span>
7983                                </p>
7984                            </a>
7985                            <a class="w3-text" href="nguyen24_interspeech.html">
7986                                <p>
7987                                    Identifying Speakers in Dialogue Transcripts: A Text-based Approach Using Pretrained Language Models
7988                                    <br>
7989                                    <span class="w3-text w3-text-theme">
7990                                        Minh Nguyen, Franck Dernoncourt, Seunghyun Yoon, Hanieh Deilamsalehy, Hao Tan, Ryan Rossi, Quan Hung Tran, Trung Bui, Thien Huu Nguyen
7991                                    </span>
7992                                </p>
7993                            </a>
7994                            <a class="w3-text" href="kc24_interspeech.html">
7995                                <p>
7996                                    Attention-augmented X-vectors for the Evaluation of Mimicked Speech Using Sparse Autoencoder-LSTM framework
7997                                    <br>
7998                                    <span class="w3-text w3-text-theme">
7999                                        Bhasi K. C., Rajeev Rajan, Noumida A
8000                                    </span>
8001                                </p>
8002                            </a>
8003                        </div>
8004                    </div>
8005                    <br>
8006                    <div class="w3-content" style="height:10px"  id="Speech and Audio Analysis"></div>
8007                    <div class="w3-card w3-round w3-white w3-padding">
8008                        <div class="w3-container"  style="margin-top:40px">
8009                            <h4 class="w3-center">Speech and Audio Analysis</h4>
8010                            <hr>
8011                            <a class="w3-text" href="almudevar24_interspeech.html">
8012                                <p>
8013                                    Predefined Prototypes for Intra-Class Separation and Disentanglement
8014                                    <br>
8015                                    <span class="w3-text w3-text-theme">
8016                                        Antonio Almudévar, Théo Mariotte, Alfonso Ortega, Marie Tahon, Luis Vicente, Antonio Miguel, Eduardo Lleida
8017                                    </span>
8018                                </p>
8019                            </a>
8020                            <a class="w3-text" href="koriyama24_interspeech.html">
8021                                <p>
8022                                    VAE-based Phoneme Alignment Using Gradient Annealing and SSL Acoustic Features
8023                                    <br>
8024                                    <span class="w3-text w3-text-theme">
8025                                        Tomoki Koriyama
8026                                    </span>
8027                                </p>
8028                            </a>
8029                            <a class="w3-text" href="karan24_interspeech.html">
8030                                <p>
8031                                    A Transformer-Based Voice Activity Detector
8032                                    <br>
8033                                    <span class="w3-text w3-text-theme">
8034                                        Biswajit Karan, Joshua Jansen van Vüren, Febe de Wet, Thomas Niesler
8035                                    </span>
8036                                </p>
8037                            </a>
8038                            <a class="w3-text" href="dumpala24_interspeech.html">
8039                                <p>
8040                                    XANE: eXplainable Acoustic Neural Embeddings
8041                                    <br>
8042                                    <span class="w3-text w3-text-theme">
8043                                        Sri Harsha Dumpala, Dushyant Sharma, Chandramouli Shama Sastry, Stanislav Kruchinin, James Fosburgh, Patrick A. Naylor
8044                                    </span>
8045                                </p>
8046                            </a>
8047                            <a class="w3-text" href="mallela24_interspeech.html">
8048                                <p>
8049                                    A comparative analysis of sequential models that integrate syllable dependency for automatic syllable stress detection
8050                                    <br>
8051                                    <span class="w3-text w3-text-theme">
8052                                        Jhansi Mallela, Sai Harshitha Aluru, Chiranjeevi Yarra
8053                                    </span>
8054                                </p>
8055                            </a>
8056                            <a class="w3-text" href="li24p_interspeech.html">
8057                                <p>
8058                                    Motion Based Audio-Visual Segmentation
8059                                    <br>
8060                                    <span class="w3-text w3-text-theme">
8061                                        Jiahao Li, Miao Liu, Shu Yang, Jing Wang, Xiang Xie
8062                                    </span>
8063                                </p>
8064                            </a>
8065                        </div>
8066                    </div>
8067                    <br>
8068                    <div class="w3-content" style="height:10px"  id="Speech Quality and Intelligibility: Prediction and Enhancement"></div>
8069                    <div class="w3-card w3-round w3-white w3-padding">
8070                        <div class="w3-container"  style="margin-top:40px">
8071                            <h4 class="w3-center">Speech Quality and Intelligibility: Prediction and Enhancement</h4>
8072                            <hr>
8073                            <a class="w3-text" href="best24_interspeech.html">
8074                                <p>
8075                                    Transfer Learning from Whisper for Microscopic Intelligibility Prediction
8076                                    <br>
8077                                    <span class="w3-text w3-text-theme">
8078                                        Paul Best, Santiago Cuervo, Ricard Marxer
8079                                    </span>
8080                                </p>
8081                            </a>
8082                            <a class="w3-text" href="zezario24_interspeech.html">
8083                                <p>
8084                                    Non-Intrusive Speech Intelligibility Prediction for Hearing Aids using Whisper and Metadata
8085                                    <br>
8086                                    <span class="w3-text w3-text-theme">
8087                                        Ryandhimas E. Zezario, Fei Chen, Chiou-Shann Fuh, Hsin-Min Wang, Yu Tsao
8088                                    </span>
8089                                </p>
8090                            </a>
8091                            <a class="w3-text" href="wang24y_interspeech.html">
8092                                <p>
8093                                    No-Reference Speech Intelligibility Prediction Leveraging a Noisy-Speech ASR Pre-Trained Model
8094                                    <br>
8095                                    <span class="w3-text w3-text-theme">
8096                                        Haolan Wang, Amin Edraki, Wai-Yip Chan, Iván López-Espejo, Jesper Jensen
8097                                    </span>
8098                                </p>
8099                            </a>
8100                            <a class="w3-text" href="deoliveira24_interspeech.html">
8101                                <p>
8102                                    The PESQetarian: On the Relevance of Goodhart's Law for Speech Enhancement
8103                                    <br>
8104                                    <span class="w3-text w3-text-theme">
8105                                        Danilo de Oliveira, Simon Welker, Julius Richter, Timo Gerkmann
8106                                    </span>
8107                                </p>
8108                            </a>
8109                            <a class="w3-text" href="ta24b_interspeech.html">
8110                                <p>
8111                                    Enhancing Non-Matching Reference Speech Quality Assessment through Dynamic Weight Adaptation
8112                                    <br>
8113                                    <span class="w3-text w3-text-theme">
8114                                        Bao Thang Ta, Van Hai Do, Huynh Thi Thanh Binh
8115                                    </span>
8116                                </p>
8117                            </a>
8118                            <a class="w3-text" href="chen24j_interspeech.html">
8119                                <p>
8120                                    Exploring Sentence Type Effects on the Lombard Effect and Intelligibility Enhancement: A Comparative Study of Natural and Grid Sentences
8121                                    <br>
8122                                    <span class="w3-text w3-text-theme">
8123                                        Hongyang Chen, Yuhong Yang, Zhongyuan Wang, Weiping Tu, Haojun Ai, Cedar Lin
8124                                    </span>
8125                                </p>
8126                            </a>
8127                        </div>
8128                    </div>
8129                    <br>
8130                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Vocoders"></div>
8131                    <div class="w3-card w3-round w3-white w3-padding">
8132                        <div class="w3-container"  style="margin-top:40px">
8133                            <h4 class="w3-center">Speech Synthesis: Vocoders</h4>
8134                            <hr>
8135                            <a class="w3-text" href="lv24_interspeech.html">
8136                                <p>
8137                                    FreeV: Free Lunch For Vocoders Through Pseudo Inversed Mel Filter
8138                                    <br>
8139                                    <span class="w3-text w3-text-theme">
8140                                        Yuanjun Lv, Hai Li, Ying Yan, Junhui Liu, Danming Xie, Lei Xie
8141                                    </span>
8142                                </p>
8143                            </a>
8144                            <a class="w3-text" href="chaudhary24b_interspeech.html">
8145                                <p>
8146                                    QGAN: Low Footprint Quaternion Neural Vocoder for Speech Synthesis
8147                                    <br>
8148                                    <span class="w3-text w3-text-theme">
8149                                        Aryan Chaudhary, Vinayak Abrol
8150                                    </span>
8151                                </p>
8152                            </a>
8153                            <a class="w3-text" href="cho24b_interspeech.html">
8154                                <p>
8155                                    JenGAN: Stacked Shifted Filters in GAN-Based Speech Synthesis
8156                                    <br>
8157                                    <span class="w3-text w3-text-theme">
8158                                        Hyunjae Cho, Junhyeok Lee, Wonbin Jung
8159                                    </span>
8160                                </p>
8161                            </a>
8162                            <a class="w3-text" href="shen24b_interspeech.html">
8163                                <p>
8164                                    FA-GAN: Artifacts-free and Phase-aware High-fidelity GAN-based Vocoder
8165                                    <br>
8166                                    <span class="w3-text w3-text-theme">
8167                                        Rubing Shen, Yanzhen Ren, Zongkun Sun
8168                                    </span>
8169                                </p>
8170                            </a>
8171                            <a class="w3-text" href="chen24x_interspeech.html">
8172                                <p>
8173                                    QHM-GAN: Neural Vocoder based on Quasi-Harmonic Modeling
8174                                    <br>
8175                                    <span class="w3-text w3-text-theme">
8176                                        Shaowen Chen, Tomoki Toda
8177                                    </span>
8178                                </p>
8179                            </a>
8180                            <a class="w3-text" href="du24_interspeech.html">
8181                                <p>
8182                                    BiVocoder: A Bidirectional Neural Vocoder Integrating Feature Extraction and Waveform Generation
8183                                    <br>
8184                                    <span class="w3-text w3-text-theme">
8185                                        Hui-Peng Du, Ye-Xin Lu, Yang Ai, Zhen-Hua Ling
8186                                    </span>
8187                                </p>
8188                            </a>
8189                        </div>
8190                    </div>
8191                    <br>
8192                    <div class="w3-content" style="height:10px"  id="ASR Model Training Methods"></div>
8193                    <div class="w3-card w3-round w3-white w3-padding">
8194                        <div class="w3-container"  style="margin-top:40px">
8195                            <h4 class="w3-center">ASR Model Training Methods</h4>
8196                            <hr>
8197                            <a class="w3-text" href="raissi24_interspeech.html">
8198                                <p>
8199                                    Investigating the Effect of Label Topology and Training Criterion on ASR Performance and Alignment Quality
8200                                    <br>
8201                                    <span class="w3-text w3-text-theme">
8202                                        Tina Raissi, Christoph Lüscher, Simon Berger, Ralf Schlüter, Hermann Ney
8203                                    </span>
8204                                </p>
8205                            </a>
8206                            <a class="w3-text" href="gaur24_interspeech.html">
8207                                <p>
8208                                    ASTRA: Aligning Speech and Text Representations for Asr without Sampling
8209                                    <br>
8210                                    <span class="w3-text w3-text-theme">
8211                                        Neeraj Gaur, Rohan Agrawal, Gary Wang, Parisa Haghani, Andrew Rosenberg, Bhuvana Ramabhadran
8212                                    </span>
8213                                </p>
8214                            </a>
8215                            <a class="w3-text" href="shakeel24_interspeech.html">
8216                                <p>
8217                                    Contextualized End-to-end Automatic Speech Recognition with Intermediate Biasing Loss
8218                                    <br>
8219                                    <span class="w3-text w3-text-theme">
8220                                        Muhammad Shakeel, Yui Sudo, Yifan Peng, Shinji Watanabe
8221                                    </span>
8222                                </p>
8223                            </a>
8224                            <a class="w3-text" href="chiu24_interspeech.html">
8225                                <p>
8226                                    Learnable Layer Selection and Model Fusion for Speech Self-Supervised Learning Models
8227                                    <br>
8228                                    <span class="w3-text w3-text-theme">
8229                                        Sheng-Chieh Chiu, Chia-Hua Wu, Jih-Kang Hsieh, Yu Tsao, Hsin-Min Wang
8230                                    </span>
8231                                </p>
8232                            </a>
8233                            <a class="w3-text" href="kulshreshtha24_interspeech.html">
8234                                <p>
8235                                    Sequential Editing for Lifelong Training of Speech Recognition Models
8236                                    <br>
8237                                    <span class="w3-text w3-text-theme">
8238                                        Devang Kulshreshtha, Nikolaos Pappas, Brady Houston, Saket Dingliwal, Srikanth Ronanki
8239                                    </span>
8240                                </p>
8241                            </a>
8242                            <a class="w3-text" href="yeh24_interspeech.html">
8243                                <p>
8244                                    Cross-Modality Diffusion Modeling and Sampling for Speech Recognition
8245                                    <br>
8246                                    <span class="w3-text w3-text-theme">
8247                                        Chia-Kai Yeh, Chih-Chun Chen, Ching-Hsien Hsu, Jen-Tzung Chien
8248                                    </span>
8249                                </p>
8250                            </a>
8251                        </div>
8252                    </div>
8253                    <br>
8254                    <div class="w3-content" style="height:10px"  id="Cross-Lingual and Multilingual Processing"></div>
8255                    <div class="w3-card w3-round w3-white w3-padding">
8256                        <div class="w3-container"  style="margin-top:40px">
8257                            <h4 class="w3-center">Cross-Lingual and Multilingual Processing</h4>
8258                            <hr>
8259                            <a class="w3-text" href="liu24l_interspeech.html">
8260                                <p>
8261                                    A Parameter-efficient Language Extension Framework for Multilingual ASR
8262                                    <br>
8263                                    <span class="w3-text w3-text-theme">
8264                                        Wei Liu, Jingyong Hou, Dong Yang, Muyong Cao, Tan Lee
8265                                    </span>
8266                                </p>
8267                            </a>
8268                            <a class="w3-text" href="song24_interspeech.html">
8269                                <p>
8270                                    LoRA-Whisper: Parameter-Efficient and Extensible Multilingual ASR
8271                                    <br>
8272                                    <span class="w3-text w3-text-theme">
8273                                        Zheshu Song, Jianheng Zhuo, Yifan Yang, Ziyang Ma, Shixiong Zhang, Xie Chen
8274                                    </span>
8275                                </p>
8276                            </a>
8277                            <a class="w3-text" href="zanonboito24_interspeech.html">
8278                                <p>
8279                                    mHuBERT-147: A Compact Multilingual HuBERT Model
8280                                    <br>
8281                                    <span class="w3-text w3-text-theme">
8282                                        Marcely Zanon Boito, Vivek Iyer, Nikolaos Lagos, Laurent Besacier, Ioan Calapodescu
8283                                    </span>
8284                                </p>
8285                            </a>
8286                            <a class="w3-text" href="lodagala24_interspeech.html">
8287                                <p>
8288                                    All Ears: Building Self-Supervised Learning based ASR models for Indian Languages at scale
8289                                    <br>
8290                                    <span class="w3-text w3-text-theme">
8291                                        Vasista Sai Lodagala, Abhishek Biswas, Shoutrik Das, Jordan F, S Umesh
8292                                    </span>
8293                                </p>
8294                            </a>
8295                            <a class="w3-text" href="jakhar24_interspeech.html">
8296                                <p>
8297                                    A Unified Approach to Multilingual Automatic Speech Recognition with Improved Language Identification for Indic Languages
8298                                    <br>
8299                                    <span class="w3-text w3-text-theme">
8300                                        Nikhil Jakhar, Sudhanshu Srivastava, Arun Baby
8301                                    </span>
8302                                </p>
8303                            </a>
8304                            <a class="w3-text" href="dong24_interspeech.html">
8305                                <p>
8306                                    Integrating Speech Self-Supervised Learning Models and Large Language Models for ASR
8307                                    <br>
8308                                    <span class="w3-text w3-text-theme">
8309                                        Ling Dong, Zhengtao Yu, Wenjun Wang, Yuxin Huang, Shengxiang Gao, Guojiang Zhou
8310                                    </span>
8311                                </p>
8312                            </a>
8313                            <a class="w3-text" href="tian24_interspeech.html">
8314                                <p>
8315                                    On the Effects of Heterogeneous Data Sources on Speech-to-Text Foundation Models
8316                                    <br>
8317                                    <span class="w3-text w3-text-theme">
8318                                        Jinchuan Tian, Yifan Peng, William Chen, Kwanghee Choi, Karen Livescu, Shinji Watanabe
8319                                    </span>
8320                                </p>
8321                            </a>
8322                            <a class="w3-text" href="puvvada24_interspeech.html">
8323                                <p>
8324                                    Less is More: Accurate Speech Recognition &amp; Translation without Web-Scale Data
8325                                    <br>
8326                                    <span class="w3-text w3-text-theme">
8327                                        Krishna C. Puvvada, Piotr Żelasko, He Huang, Oleksii Hrinchuk, Nithin Rao Koluguri, Kunal Dhawan, Somshubra Majumdar, Elena Rastorgueva, Zhehuai Chen, Vitaly Lavrukhin, Jagadeesh Balam, Boris Ginsburg
8328                                    </span>
8329                                </p>
8330                            </a>
8331                            <a class="w3-text" href="paraskevopoulos24_interspeech.html">
8332                                <p>
8333                                    The Greek podcast corpus: Competitive speech models for low-resourced languages with weakly supervised data
8334                                    <br>
8335                                    <span class="w3-text w3-text-theme">
8336                                        Georgios Paraskevopoulos, Chara Tsoukala, Athanasios Katsamanis, Vassilis Katsouros
8337                                    </span>
8338                                </p>
8339                            </a>
8340                            <a class="w3-text" href="vakirtzian24_interspeech.html">
8341                                <p>
8342                                    Speech Recognition for Greek Dialects: A Challenging Benchmark
8343                                    <br>
8344                                    <span class="w3-text w3-text-theme">
8345                                        Socrates Vakirtzian, Chara Tsoukala, Stavros Bompolas, Katerina Mouzou, Vivian Stamou, Georgios Paraskevopoulos, Antonios Dimakis, Stella Markantonatou, Angela Ralli, Antonios Anastasopoulos
8346                                    </span>
8347                                </p>
8348                            </a>
8349                            <a class="w3-text" href="liu24k_interspeech.html">
8350                                <p>
8351                                    LUPET: Incorporating Hierarchical Information Path into Multilingual ASR
8352                                    <br>
8353                                    <span class="w3-text w3-text-theme">
8354                                        Wei Liu, Jingyong Hou, Dong Yang, Muyong Cao, Tan Lee
8355                                    </span>
8356                                </p>
8357                            </a>
8358                            <a class="w3-text" href="kummervold24_interspeech.html">
8359                                <p>
8360                                    Whispering in Norwegian: Navigating Orthographic and Dialectic Challenges
8361                                    <br>
8362                                    <span class="w3-text w3-text-theme">
8363                                        Per E Kummervold, Javier de la Rosa, Freddy Wetjen, Rolv-Arild Braaten, Per Erik Solberg
8364                                    </span>
8365                                </p>
8366                            </a>
8367                            <a class="w3-text" href="srivastava24_interspeech.html">
8368                                <p>
8369                                    EFFUSE: Efficient Self-Supervised Feature Fusion for E2E ASR in Low Resource and Multilingual Scenarios
8370                                    <br>
8371                                    <span class="w3-text w3-text-theme">
8372                                        Tejes Srivastava, Jiatong Shi, William Chen, Shinji Watanabe
8373                                    </span>
8374                                </p>
8375                            </a>
8376                            <a class="w3-text" href="hussein24_interspeech.html">
8377                                <p>
8378                                    Enhancing Neural Transducer for Multilingual ASR with Synchronized Language Diarization
8379                                    <br>
8380                                    <span class="w3-text w3-text-theme">
8381                                        Amir Hussein, Desh Raj, Matthew Wiesner, Daniel Povey, Paola Garcia, Sanjeev Khudanpur
8382                                    </span>
8383                                </p>
8384                            </a>
8385                            <a class="w3-text" href="ye24_interspeech.html">
8386                                <p>
8387                                    SC-MoE: Switch Conformer Mixture of Experts for Unified Streaming and Non-streaming Code-Switching ASR
8388                                    <br>
8389                                    <span class="w3-text w3-text-theme">
8390                                        Shuaishuai Ye, Shunfei Chen, Xinhui Hu, Xinkang Xu
8391                                    </span>
8392                                </p>
8393                            </a>
8394                        </div>
8395                    </div>
8396                    <br>
8397                    <div class="w3-content" style="height:10px"  id="Speech Assessment"></div>
8398                    <div class="w3-card w3-round w3-white w3-padding">
8399                        <div class="w3-container"  style="margin-top:40px">
8400                            <h4 class="w3-center">Speech Assessment</h4>
8401                            <hr>
8402                            <a class="w3-text" href="wu24i_interspeech.html">
8403                                <p>
8404                                    Optimizing Automatic Speech Assessment: W-RankSim Regularization and Hybrid Feature Fusion Strategies
8405                                    <br>
8406                                    <span class="w3-text w3-text-theme">
8407                                        Chung-Wen Wu, Berlin Chen
8408                                    </span>
8409                                </p>
8410                            </a>
8411                            <a class="w3-text" href="cheng24b_interspeech.html">
8412                                <p>
8413                                    Context-Aware Speech Recognition Using Prompts for Language Learners
8414                                    <br>
8415                                    <span class="w3-text w3-text-theme">
8416                                        Jian Cheng
8417                                    </span>
8418                                </p>
8419                            </a>
8420                            <a class="w3-text" href="gothi24_interspeech.html">
8421                                <p>
8422                                    A Dataset and Two-pass System for Reading Miscue Detection
8423                                    <br>
8424                                    <span class="w3-text w3-text-theme">
8425                                        Raj Gothi, Rahul Kumar, Mildred Pereira, Nagesh Nayak, Preeti Rao
8426                                    </span>
8427                                </p>
8428                            </a>
8429                            <a class="w3-text" href="lun24_interspeech.html">
8430                                <p>
8431                                    Oversampling, Augmentation and Curriculum Learning for Speaking Assessment with Limited Training Data
8432                                    <br>
8433                                    <span class="w3-text w3-text-theme">
8434                                        Tin Mei Lun, Ekaterina Voskoboinik, Ragheb Al-Ghezi, Tamas Grosz, Mikko Kurimo
8435                                    </span>
8436                                </p>
8437                            </a>
8438                            <a class="w3-text" href="tomita24_interspeech.html">
8439                                <p>
8440                                    Analysis and Visualization of Directional Diversity in Listening Fluency of World Englishes Speakers in the Framework of Mutual Shadowing
8441                                    <br>
8442                                    <span class="w3-text w3-text-theme">
8443                                        Yu Tomita, Yingxiang Gao, Nobuaki Minematsu, Noriko Nakanishi, Daisuke Saito
8444                                    </span>
8445                                </p>
8446                            </a>
8447                            <a class="w3-text" href="robertson24_interspeech.html">
8448                                <p>
8449                                    Quantifying the Role of Textual Predictability in Automatic Speech Recognition
8450                                    <br>
8451                                    <span class="w3-text w3-text-theme">
8452                                        Sean Robertson, Gerald Penn, Ewan Dunbar
8453                                    </span>
8454                                </p>
8455                            </a>
8456                        </div>
8457                    </div>
8458                    <br>
8459                    <div class="w3-content" style="height:10px"  id="Question Answering from Speech and Spoken Dialogue Systems"></div>
8460                    <div class="w3-card w3-round w3-white w3-padding">
8461                        <div class="w3-container"  style="margin-top:40px">
8462                            <h4 class="w3-center">Question Answering from Speech and Spoken Dialogue Systems</h4>
8463                            <hr>
8464                            <a class="w3-text" href="rajkhowa24_interspeech.html">
8465                                <p>
8466                                    TM-PATHVQA: 90000+ Textless Multilingual Questions for Medical Visual Question Answering
8467                                    <br>
8468                                    <span class="w3-text w3-text-theme">
8469                                        Tonmoy Rajkhowa, Amartya Roy Chowdhury, Sankalp Nagaonkar, Achyut Mani Tripathi, Mahadeva Prasanna
8470                                    </span>
8471                                </p>
8472                            </a>
8473                            <a class="w3-text" href="phukan24_interspeech.html">
8474                                <p>
8475                                    Towards Multilingual Audio-Visual Question Answering
8476                                    <br>
8477                                    <span class="w3-text w3-text-theme">
8478                                        Orchid Chetia Phukan, Priyabrata Mallick, Swarup Ranjan Behera, Aalekhya Satya Narayani, Arun Balaji Buduru, Rajesh Sharma
8479                                    </span>
8480                                </p>
8481                            </a>
8482                            <a class="w3-text" href="nguyen24c_interspeech.html">
8483                                <p>
8484                                    Reinforcement Learning from Answer Reranking Feedback for Retrieval-Augmented Answer Generation
8485                                    <br>
8486                                    <span class="w3-text w3-text-theme">
8487                                        Minh Nguyen, Toan Quoc Nguyen, Kishan KC, Zeyu Zhang, Thuy Vu
8488                                    </span>
8489                                </p>
8490                            </a>
8491                            <a class="w3-text" href="noroozi24_interspeech.html">
8492                                <p>
8493                                    Instruction Data Generation and Unsupervised Adaptation for Speech Language Models
8494                                    <br>
8495                                    <span class="w3-text w3-text-theme">
8496                                        Vahid Noroozi, Zhehuai Chen, Somshubra Majumdar, Steve Huang, Jagadeesh Balam, Boris Ginsburg
8497                                    </span>
8498                                </p>
8499                            </a>
8500                            <a class="w3-text" href="dibratto24_interspeech.html">
8501                                <p>
8502                                    On the Use of Plausible Arguments in Explainable Conversational AI
8503                                    <br>
8504                                    <span class="w3-text w3-text-theme">
8505                                        Martina Di Bratto, Maria Di Maro, Antonio Origlia
8506                                    </span>
8507                                </p>
8508                            </a>
8509                            <a class="w3-text" href="baihaqi24_interspeech.html">
8510                                <p>
8511                                    Rapport-Driven Virtual Agent: Rapport Building Dialogue Strategy for Improving User Experience at First Meeting
8512                                    <br>
8513                                    <span class="w3-text w3-text-theme">
8514                                        Muhammad Yeza Baihaqi, Angel Garcia Contreras, Seiya Kawano, Koichiro Yoshino
8515                                    </span>
8516                                </p>
8517                            </a>
8518                            <a class="w3-text" href="zhou24c_interspeech.html">
8519                                <p>
8520                                    Cross-Modal Denoising: A Novel Training Paradigm for Enhancing Speech-Image Retrieval
8521                                    <br>
8522                                    <span class="w3-text w3-text-theme">
8523                                        Lifeng Zhou, Yuke Li, Rui Deng, Yuting Yang, Haoqi Zhu
8524                                    </span>
8525                                </p>
8526                            </a>
8527                        </div>
8528                    </div>
8529                    <br>
8530                    <div class="w3-content" style="height:10px"  id="Spoken Dialogue Systems and Conversational Analysis 3"></div>
8531                    <div class="w3-card w3-round w3-white w3-padding">
8532                        <div class="w3-container"  style="margin-top:40px">
8533                            <h4 class="w3-center">Spoken Dialogue Systems and Conversational Analysis 3</h4>
8534                            <hr>
8535                            <a class="w3-text" href="huang24b_interspeech.html">
8536                                <p>
8537                                    MM-NodeFormer: Node Transformer Multimodal Fusion for Emotion Recognition in Conversation
8538                                    <br>
8539                                    <span class="w3-text w3-text-theme">
8540                                        Zilong Huang, Man-Wai Mak, Kong Aik Lee
8541                                    </span>
8542                                </p>
8543                            </a>
8544                            <a class="w3-text" href="shi24d_interspeech.html">
8545                                <p>
8546                                    Emotional Cues Extraction and Fusion for Multi-modal Emotion Prediction and Recognition in Conversation
8547                                    <br>
8548                                    <span class="w3-text w3-text-theme">
8549                                        Haoxiang Shi, Ziqi Liang, Jun Yu
8550                                    </span>
8551                                </p>
8552                            </a>
8553                            <a class="w3-text" href="suzuki24_interspeech.html">
8554                                <p>
8555                                    Participant-Pair-Wise Bottleneck Transformer for Engagement Estimation from Video Conversation
8556                                    <br>
8557                                    <span class="w3-text w3-text-theme">
8558                                        Keita Suzuki, Nobukatsu Hojo, Kazutoshi Shinoda, Saki Mizuno, Ryo Masumura
8559                                    </span>
8560                                </p>
8561                            </a>
8562                            <a class="w3-text" href="omahony24_interspeech.html">
8563                                <p>
8564                                    Well, what can you do with messy data? Exploring the prosody and pragmatic function of the discourse marker &quot;well&quot; with found data and speech synthesis
8565                                    <br>
8566                                    <span class="w3-text w3-text-theme">
8567                                        Johannah O'Mahony, Catherine Lai, Éva Székely
8568                                    </span>
8569                                </p>
8570                            </a>
8571                            <a class="w3-text" href="shinoda24_interspeech.html">
8572                                <p>
8573                                    Learning from Multiple Annotator Biased Labels in Multimodal Conversation
8574                                    <br>
8575                                    <span class="w3-text w3-text-theme">
8576                                        Kazutoshi Shinoda, Nobukatsu Hojo, Saki Mizuno, Keita Suzuki, Satoshi Kobashikawa, Ryo Masumura
8577                                    </span>
8578                                </p>
8579                            </a>
8580                            <a class="w3-text" href="hoscilowicz24_interspeech.html">
8581                                <p>
8582                                    Non-Linear Inference Time Intervention: Improving LLM Truthfulness
8583                                    <br>
8584                                    <span class="w3-text w3-text-theme">
8585                                        Jakub Hoscilowicz, Adam Wiacek, Jan Chojnacki, Adam Cieslak, Leszek Michon, Artur Janicki
8586                                    </span>
8587                                </p>
8588                            </a>
8589                            <a class="w3-text" href="liu24c_interspeech.html">
8590                                <p>
8591                                    Evaluating Speech Recognition Performance Towards Large Language Model Based Voice Assistants
8592                                    <br>
8593                                    <span class="w3-text w3-text-theme">
8594                                        Zhe Liu, Suyoun Kim, Ozlem Kalinli
8595                                    </span>
8596                                </p>
8597                            </a>
8598                        </div>
8599                    </div>
8600                    <br>
8601                    <div class="w3-content" style="height:10px"  id="Dysarthric Speech Assessment"></div>
8602                    <div class="w3-card w3-round w3-white w3-padding">
8603                        <div class="w3-container"  style="margin-top:40px">
8604                            <h4 class="w3-center">Dysarthric Speech Assessment</h4>
8605                            <hr>
8606                            <a class="w3-text" href="zaheera24_interspeech.html">
8607                                <p>
8608                                    Automatic Assessment of Dysarthria using Speech and synthetically generated Electroglottograph signal
8609                                    <br>
8610                                    <span class="w3-text w3-text-theme">
8611                                        Fathima Zaheera, Supritha Shetty, Gayadhar Pradhan, Deepak K T
8612                                    </span>
8613                                </p>
8614                            </a>
8615                            <a class="w3-text" href="wan24b_interspeech.html">
8616                                <p>
8617                                    CDSD: Chinese Dysarthria Speech Database
8618                                    <br>
8619                                    <span class="w3-text w3-text-theme">
8620                                        Yan Wan, Mengyi Sun, Xinchen Kang, Jingting Li, Pengfei Guo, Ming Gao, Su-Jing Wang
8621                                    </span>
8622                                </p>
8623                            </a>
8624                            <a class="w3-text" href="samptur24_interspeech.html">
8625                                <p>
8626                                    Exploring Syllable Discriminability during Diadochokinetic Task with Increasing Dysarthria Severity for Patients with Amyotrophic Lateral Sclerosis
8627                                    <br>
8628                                    <span class="w3-text w3-text-theme">
8629                                        Neelesh Samptur, Tanuka Bhattacharjee, Anirudh Chakravarty K, Seena Vengalil, Yamini Belur, Atchayaram Nalini, Prasanta Kumar Ghosh
8630                                    </span>
8631                                </p>
8632                            </a>
8633                            <a class="w3-text" href="perez24_interspeech.html">
8634                                <p>
8635                                    Beyond Binary: Multiclass Paraphasia Detection with Generative Pretrained Transformers and End-to-End Models
8636                                    <br>
8637                                    <span class="w3-text w3-text-theme">
8638                                        Matthew Perez, Aneesha Sampath, Minxue Niu, Emily Mower Provost
8639                                    </span>
8640                                </p>
8641                            </a>
8642                            <a class="w3-text" href="daoudi24_interspeech.html">
8643                                <p>
8644                                    Electroglottography for the assessment of dysphonia in Parkinson's disease and multiple system atrophy
8645                                    <br>
8646                                    <span class="w3-text w3-text-theme">
8647                                        Khalid Daoudi, Solange Milhé de Saint Victor, Alexandra Foubert-Samier, Margherita Fabbri, Anne Pavy-Le Traon, Olivier Rascol, Virginie Woisard, Wassilios G. Me
8647issner
8648                                    </span>
8649                                </p>
8650                            </a>
8651                            <a class="w3-text" href="chen24t_interspeech.html">
8652                                <p>
8653                                    CoLM-DSR: Leveraging Neural Codec Language Modeling for Multi-Modal Dysarthric Speech Reconstruction
8654                                    <br>
8655                                    <span class="w3-text w3-text-theme">
8656                                        Xueyuan Chen, Dongchao Yang, Dingdong Wang, Xixin Wu, Zhiyong Wu, Helen Meng
8657                                    </span>
8658                                </p>
8659                            </a>
8660                        </div>
8661                    </div>
8662                    <br>
8663                    <div class="w3-content" style="height:10px"  id="Spoken Language Models for Universal Speech Processing (Special Session)"></div>
8664                    <div class="w3-card w3-round w3-white w3-padding">
8665                        <div class="w3-container"  style="margin-top:40px">
8666                            <h4 class="w3-center">Spoken Language Models for Universal Speech Processing (Special Session)</h4>
8667                            <hr>
8668                            <a class="w3-text" href="li24qa_interspeech.html">
8669                                <p>
8670                                    On the Effectiveness of Acoustic BPE in Decoder-Only TTS
8671                                    <br>
8672                                    <span class="w3-text w3-text-theme">
8673                                        Bohan Li, Feiyu Shen, Yiwei Guo, Shuai Wang, Xie Chen, Kai Yu
8674                                    </span>
8675                                </p>
8676                            </a>
8677                            <a class="w3-text" href="chang24c_interspeech.html">
8678                                <p>
8679                                    Exploring In-Context Learning of Textless Speech Language Model for Speech Classification Tasks
8680                                    <br>
8681                                    <span class="w3-text w3-text-theme">
8682                                        Kai-Wei Chang, Ming-Hao Hsu, Shan-Wen Li, Hung-yi Lee
8683                                    </span>
8684                                </p>
8685                            </a>
8686                            <a class="w3-text" href="kuan24_interspeech.html">
8687                                <p>
8688                                    Understanding Sounds, Missing the Questions: The Challenge of Object Hallucination in Large Audio-Language Models
8689                                    <br>
8690                                    <span class="w3-text w3-text-theme">
8691                                        Chun-Yi Kuan, Wei-Ping Huang, Hung-yi Lee
8692                                    </span>
8693                                </p>
8694                            </a>
8695                            <a class="w3-text" href="tang24d_interspeech.html">
8696                                <p>
8697                                    Can Large Language Models Understand Spatial Audio?
8698                                    <br>
8699                                    <span class="w3-text w3-text-theme">
8700                                        Changli Tang, Wenyi Yu, Guangzhi Sun, Xianzhao Chen, Tian Tan, Wei Li, Jun Zhang, Lu Lu, Zejun Ma, Yuxuan Wang, Chao Zhang
8701                                    </span>
8702                                </p>
8703                            </a>
8704                            <a class="w3-text" href="shon24_interspeech.html">
8705                                <p>
8706                                    DiscreteSLU: A Large Language Model with Self-Supervised Discrete Speech Units for Spoken Language Understanding
8707                                    <br>
8708                                    <span class="w3-text w3-text-theme">
8709                                        Suwon Shon, Kwangyoun Kim, Yi-Te Hsu, Prashant Sridhar, Shinji Watanabe, Karen Livescu
8710                                    </span>
8711                                </p>
8712                            </a>
8713                            <a class="w3-text" href="lu24c_interspeech.html">
8714                                <p>
8715                                    DeSTA: Enhancing Speech Language Models through Descriptive Speech-Text Alignment
8716                                    <br>
8717                                    <span class="w3-text w3-text-theme">
8718                                        Ke-Han Lu, Zhehuai Chen, Szu-Wei Fu, He Huang, Boris Ginsburg, Yu-Chiang Frank Wang, Hung-yi Lee
8719                                    </span>
8720                                </p>
8721                            </a>
8722                            <a class="w3-text" href="pan24b_interspeech.html">
8723                                <p>
8724                                    COSMIC: Data Efficient Instruction-tuning For Speech In-Context Learning
8725                                    <br>
8726                                    <span class="w3-text w3-text-theme">
8727                                        Jing Pan, Jian Wu, Yashesh Gaur, Sunit Sivasankaran, Zhuo Chen, Shujie Liu, Jinyu Li
8728                                    </span>
8729                                </p>
8730                            </a>
8731                            <a class="w3-text" href="messica24_interspeech.html">
8732                                <p>
8733                                    NAST: Noise Aware Speech Tokenization for Speech Language Models
8734                                    <br>
8735                                    <span class="w3-text w3-text-theme">
8736                                        Shoval Messica, Yossi Adi
8737                                    </span>
8738                                </p>
8739                            </a>
8740                            <a class="w3-text" href="shechtman24_interspeech.html">
8741                                <p>
8742                                    Low Bitrate High-Quality RVQGAN-based Discrete Speech Tokenizer
8743                                    <br>
8744                                    <span class="w3-text w3-text-theme">
8745                                        Slava Shechtman, Avihu Dekel
8746                                    </span>
8747                                </p>
8748                            </a>
8749                        </div>
8750                    </div>
8751                    <br>
8752                    <div class="w3-content" style="height:10px"  id="Keynote 4"></div>
8753                    <div class="w3-card w3-round w3-white w3-padding">
8754                        <div class="w3-container"  style="margin-top:40px">
8755                            <h4 class="w3-center">Keynote 4</h4>
8756                            <hr>
8757                            <a class="w3-text" href="tillmann24_interspeech.html">
8758                                <p>
8759                                    Perception of music and speech: Focus on rhythm processing
8760                                    <br>
8761                                    <span class="w3-text w3-text-theme">
8762                                        Barbara Tillmann
8763                                    </span>
8764                                </p>
8765                            </a>
8766                        </div>
8767                    </div>
8768                    <br>
8769                    <div class="w3-content" style="height:10px"  id="L1/L2 Acquisition and Cross-Linguistic Factors"></div>
8770                    <div class="w3-card w3-round w3-white w3-padding">
8771                        <div class="w3-container"  style="margin-top:40px">
8772                            <h4 class="w3-center">L1/L2 Acquisition and Cross-Linguistic Factors</h4>
8773                            <hr>
8774                            <a class="w3-text" href="hwang24b_interspeech.html">
8775                                <p>
8776                                    Acquisition of high vowel devoicing in Japanese: A production experiment with three and four year olds
8777                                    <br>
8778                                    <span class="w3-text w3-text-theme">
8779                                        Hyun Kyung Hwang, Manami Hirayama
8780                                    </span>
8781                                </p>
8782                            </a>
8783                            <a class="w3-text" href="li24fa_interspeech.html">
8784                                <p>
8785                                    The Production of Contrastive Focus by  7 to 13-year-olds Learning Mandarin Chinese
8786                                    <br>
8787                                    <span class="w3-text w3-text-theme">
8788                                        Zimeng Li, Zhongxuan Mao, Shengting Shen, Ivan Yuen, Ping Tang
8789                                    </span>
8790                                </p>
8791                            </a>
8792                            <a class="w3-text" href="zaitova24_interspeech.html">
8793                                <p>
8794                                    Cross-Linguistic Intelligibility of Non-Compositional Expressions in Spoken Context
8795                                    <br>
8796                                    <span class="w3-text w3-text-theme">
8797                                        Iuliia Zaitova, Irina Stenger, Wei Xue, Tania Avgustinova, Bernd Möbius, Dietrich Klakow
8798                                    </span>
8799                                </p>
8800                            </a>
8801                            <a class="w3-text" href="demaere24_interspeech.html">
8802                                <p>
8803                                    On the relationship between speech production and vocabulary size in 3-5 year olds
8804                                    <br>
8805                                    <span class="w3-text w3-text-theme">
8806                                        Alexis DeMaere, Nicole van Rootselaar, Fangfang Li, Robbin Gibb, Claudia L. R. Gonzalez
8807                                    </span>
8808                                </p>
8809                            </a>
8810                            <a class="w3-text" href="polzehl24_interspeech.html">
8811                                <p>
8812                                    Towards Classifying Mother Tongue from Infant Cries - Findings Substantiating Prenatal Learning Theory
8813                                    <br>
8814                                    <span class="w3-text w3-text-theme">
8815                                        Tim Polzehl, Tim Herzig, Friedrich Wicke, Kathleen Wermke, Razieh Khamsehashari, Michiko Dahlem, Sebastian Möller
8816                                    </span>
8817                                </p>
8818                            </a>
8819                            <a class="w3-text" href="li24n_interspeech.html">
8820                                <p>
8821                                    Effect of Complex Boundary Tones on Tone Identification: An Experimental Study with Mandarin-speaking Preschool Children
8822                                    <br>
8823                                    <span class="w3-text w3-text-theme">
8824                                        Aijun Li, Jun Gao, Zhiwei Wang
8825                                    </span>
8826                                </p>
8827                            </a>
8828                            <a class="w3-text" href="truong24_interspeech.html">
8829                                <p>
8830                                    Ethnolinguistic Identification of Vietnamese-German Heritage Speech
8831                                    <br>
8832                                    <span class="w3-text w3-text-theme">
8833                                        Thanh Lan Truong, Andrea Weber
8834                                    </span>
8835                                </p>
8836                            </a>
8837                        </div>
8838                    </div>
8839                    <br>
8840                    <div class="w3-content" style="height:10px"  id="Speaker Stance, Emotion and Language-External Factors"></div>
8841                    <div class="w3-card w3-round w3-white w3-padding">
8842                        <div class="w3-container"  style="margin-top:40px">
8843                            <h4 class="w3-center">Speaker Stance, Emotion and Language-External Factors</h4>
8844                            <hr>
8845                            <a class="w3-text" href="hoffner24_interspeech.html">
8846                                <p>
8847                                    Joint prediction of subjective listening effort and speech intelligibility based on end-to-end learning
8848                                    <br>
8849                                    <span class="w3-text w3-text-theme">
8850                                        Dirk Eike Hoffner, Jana Roßbach, Bernd T. Meyer
8851                                    </span>
8852                                </p>
8853                            </a>
8854                            <a class="w3-text" href="wu24j_interspeech.html">
8855                                <p>
8856                                    Depression Enhances Internal Inconsistency between Spoken and Semantic Emotion: Evidence from the Analysis of Emotion Expression in Conversation
8857                                    <br>
8858                                    <span class="w3-text w3-text-theme">
8859                                        Xinyi Wu, Changqing Xu, Nan Li, Rongfeng Su, Lan Wang, Nan Yan
8860                                    </span>
8861                                </p>
8862                            </a>
8863                            <a class="w3-text" href="simantiraki24_interspeech.html">
8864                                <p>
8865                                    Listeners' F0 preferences in quiet and stationary noise
8866                                    <br>
8867                                    <span class="w3-text w3-text-theme">
8868                                        Olympia Simantiraki, Martin Cooke
8869                                    </span>
8870                                </p>
8871                            </a>
8872                            <a class="w3-text" href="hodoshima24_interspeech.html">
8873                                <p>
8874                                    Effects of talker and playback rate of reverberation-induced speech on speech intelligibility of older adults
8875                                    <br>
8876                                    <span class="w3-text w3-text-theme">
8877                                        Nao Hodoshima
8878                                    </span>
8879                                </p>
8880                            </a>
8881                        </div>
8882                    </div>
8883                    <br>
8884                    <div class="w3-content" style="height:10px"  id="Experimental Phonetics and Laboratory Phonology"></div>
8885                    <div class="w3-card w3-round w3-white w3-padding">
8886                        <div class="w3-container"  style="margin-top:40px">
8887                            <h4 class="w3-center">Experimental Phonetics and Laboratory Phonology</h4>
8888                            <hr>
8889                            <a class="w3-text" href="zhao24i_interspeech.html">
8890                                <p>
8891                                    Age-related Differences in Acoustic Cues for the Perception of Checked Syllables in Shengzhou Wu
8892                                    <br>
8893                                    <span class="w3-text w3-text-theme">
8894                                        Bingliang Zhao, Jiangping Kong, Xiyu Wu
8895                                    </span>
8896                                </p>
8897                            </a>
8898                            <a class="w3-text" href="kaland24b_interspeech.html">
8899                                <p>
8900                                    Quantity-sensitivity affects recall performance of word stress
8901                                    <br>
8902                                    <span class="w3-text w3-text-theme">
8903                                        Constantijn Kaland, Maria Lialiou
8904                                    </span>
8905                                </p>
8906                            </a>
8907                            <a class="w3-text" href="tokac24_interspeech.html">
8908                                <p>
8909                                    Phonological Symmetry Does Not Predict Generalization of Perceptual Adaptation to Vowels
8910                                    <br>
8911                                    <span class="w3-text w3-text-theme">
8912                                        Zuheyra Tokac, Jennifer Cole
8913                                    </span>
8914                                </p>
8915                            </a>
8916                            <a class="w3-text" href="reitsema24_interspeech.html">
8917                                <p>
8918                                    Perceptual Learning in Lexical Tone: Phonetic Similarity vs. Phonological Categories
8919                                    <br>
8920                                    <span class="w3-text w3-text-theme">
8921                                        Ariëlle Reitsema, Chenxin Li, Leanne van Lambalgen, Laura Preining, Saskia Galindo Jong, Qing Yang, Xinyi Wen, Yiya Chen
8922                                    </span>
8923                                </p>
8924                            </a>
8925                            <a class="w3-text" href="stein24_interspeech.html">
8926                                <p>
8927                                    Modeling probabilistic reduction across domains with Naive Discriminative Learning
8928                                    <br>
8929                                    <span class="w3-text w3-text-theme">
8930                                        Anna Stein, Kevin Tang
8931                                    </span>
8932                                </p>
8933                            </a>
8934                            <a class="w3-text" href="lee24l_interspeech.html">
8935                                <p>
8936                                    Do we EXPECT TO find phonetic traces for syntactic traces?
8937                                    <br>
8938                                    <span class="w3-text w3-text-theme">
8939                                        Jonathan Him Nok Lee, Mark Liberman, Martin Salzmann
8940                                    </span>
8941                                </p>
8942                            </a>
8943                        </div>
8944                    </div>
8945                    <br>
8946                    <div class="w3-content" style="height:10px"  id="Speaker recognition evaluation and resources"></div>
8947                    <div class="w3-card w3-round w3-white w3-padding">
8948                        <div class="w3-container"  style="margin-top:40px">
8949                            <h4 class="w3-center">Speaker recognition evaluation and resources</h4>
8950                            <hr>
8951                            <a class="w3-text" href="lin24j_interspeech.html">
8952                                <p>
8953                                    VoxBlink2: A 100K+ Speaker Recognition Corpus and the Open-Set Speaker-Identification Benchmark
8954                                    <br>
8955                                    <span class="w3-text w3-text-theme">
8956                                        Yuke Lin, Ming Cheng, Fulin Zhang, Yingying Gao, Shilei Zhang, Ming Li
8957                                    </span>
8958                                </p>
8959                            </a>
8960                            <a class="w3-text" href="hutiri24_interspeech.html">
8961                                <p>
8962                                    As Biased as You Measure: Methodological Pitfalls of Bias Evaluations in Speaker Verification Research
8963                                    <br>
8964                                    <span class="w3-text w3-text-theme">
8965                                        Wiebke Hutiri, Tanvina Patel, Aaron Yi Ding, Odette Scharenborg
8966                                    </span>
8967                                </p>
8968                            </a>
8969                            <a class="w3-text" href="wang24fa_interspeech.html">
8970                                <p>
8971                                    WeSep: A Scalable and Flexible Toolkit Towards Generalizable Target Speaker Extraction
8972                                    <br>
8973                                    <span class="w3-text w3-text-theme">
8974                                        Shuai Wang, Ke Zhang, Shaoxiong Lin, Junjie Li, Xuefei Wang, Meng Ge, Jianwei Yu, Yanmin Qian, Haizhou Li
8975                                    </span>
8976                                </p>
8977                            </a>
8978                            <a class="w3-text" href="jung24c_interspeech.html">
8979                                <p>
8980                                    ESPnet-SPK: full pipeline speaker embedding toolkit with reproducible recipes, self-supervised front-ends, and off-the-shelf models
8981                                    <br>
8982                                    <span class="w3-text w3-text-theme">
8983                                        Jee-weon Jung, Wangyou Zhang, Jiatong Shi, Zakaria Aldeneh, Takuya Higuchi, Alex Gichamba, Barry-John Theobald, Ahmed Hussen Abdelaziz, Shinji Watanabe
8984                                    </span>
8985                                </p>
8986                            </a>
8987                            <a class="w3-text" href="huang24g_interspeech.html">
8988                                <p>
8989                                    Active Speaker Detection in Fisheye Meeting Scenes with Scene Spatial Spectrums
8990                                    <br>
8991                                    <span class="w3-text w3-text-theme">
8992                                        Xinghao Huang, Weiwei Jiang, Long Rao, Wei Xu, Wenqing Cheng
8993                                    </span>
8994                                </p>
8995                            </a>
8996                            <a class="w3-text" href="hoang24b_interspeech.html">
8997                                <p>
8998                                    VSASV: a Vietnamese Dataset for Spoofing-Aware Speaker Verification
8999                                    <br>
9000                                    <span class="w3-text w3-text-theme">
9001                                        Vu Hoang, Viet Thanh Pham, Hoa Nguyen Xuan, Pham Nhi, Phuong Dat, Thi Thu Trang Nguyen
9002                                    </span>
9003                                </p>
9004                            </a>
9005                        </div>
9006                    </div>
9007                    <br>
9008                    <div class="w3-content" style="height:10px"  id="Speech Type Classification"></div>
9009                    <div class="w3-card w3-round w3-white w3-padding">
9010                        <div class="w3-container"  style="margin-top:40px">
9011                            <h4 class="w3-center">Speech Type Classification</h4>
9012                            <hr>
9013                            <a class="w3-text" href="ma24_interspeech.html">
9014                                <p>
9015                                    E-ODN: An Emotion Open Deep Network for Generalised and Adaptive Speech Emotion Recognition
9016                                    <br>
9017                                    <span class="w3-text w3-text-theme">
9018                                        Liuxian Ma, Lin Shen, Ruobing Li, Haojie Zhang, Kun Qian, Bin Hu, Björn W. Schuller, Yoshiharu Yamamoto
9019                                    </span>
9020                                </p>
9021                            </a>
9022                            <a class="w3-text" href="liu24h_interspeech.html">
9023                                <p>
9024                                    Enhancing Multilingual Voice Toxicity Detection with Speech-Text Alignment
9025                                    <br>
9026                                    <span class="w3-text w3-text-theme">
9027                                        Joseph Liu, Mahesh Kumar Nandwana, Janne Pylkkönen, Hannes Heikinheimo, Morgan McGuire
9028                                    </span>
9029                                </p>
9030                            </a>
9031                            <a class="w3-text" href="nafea24_interspeech.html">
9032                                <p>
9033                                    AraOffence: Detecting Offensive Speech Across Dialects in Arabic Media
9034                                    <br>
9035                                    <span class="w3-text w3-text-theme">
9036                                        Youssef Nafea, Shady Shehata, Zeerak Talat, Ahmed Aboeitta, Ahmed Sharshar, Preslav Nakov
9037                                    </span>
9038                                </p>
9039                            </a>
9040                            <a class="w3-text" href="cheng24c_interspeech.html">
9041                                <p>
9042                                    CogniVoice: Multimodal and Multilingual Fusion Networks for Mild Cognitive Impairment Assessment from Spontaneous Speech
9043                                    <br>
9044                                    <span class="w3-text w3-text-theme">
9045                                        Jiali Cheng, Mohamed Elgaar, Nidhi Vakil, Hadi Amiri
9046                                    </span>
9047                                </p>
9048                            </a>
9049                            <a class="w3-text" href="niu24b_interspeech.html">
9050                                <p>
9051                                    Speech Topic Classification Based on Multi-Scale and Graph Attention Networks
9052                                    <br>
9053                                    <span class="w3-text w3-text-theme">
9054                                        Fangjing Niu, Xiaozhe Qi, Xinya Chen, Liang He
9055                                    </span>
9056                                </p>
9057                            </a>
9058                            <a class="w3-text" href="chen24r_interspeech.html">
9059                                <p>
9060                                    Enhancing Speech and Music Discrimination Through the Integration of Static and Dynamic Features
9061                                    <br>
9062                                    <span class="w3-text w3-text-theme">
9063                                        Liangwei Chen, Xiren Zhou, Qiang Tu, Huanhuan Chen
9064                                    </span>
9065                                </p>
9066                            </a>
9067                        </div>
9068                    </div>
9069                    <br>
9070                    <div class="w3-content" style="height:10px"  id="Target Speaker Extraction"></div>
9071                    <div class="w3-card w3-round w3-white w3-padding">
9072                        <div class="w3-container"  style="margin-top:40px">
9073                            <h4 class="w3-center">Target Speaker Extraction</h4>
9074                            <hr>
9075                            <a class="w3-text" href="meng24b_interspeech.html">
9076                                <p>
9077                                    Binaural Selective Attention Model for Target Speaker Extraction
9078                                    <br>
9079                                    <span class="w3-text w3-text-theme">
9080                                        Hanyu Meng, Qiquan Zhang, Xiangyu Zhang, Vidhyasaharan Sethu, Eliathamby Ambikairajah
9081                                    </span>
9082                                </p>
9083                            </a>
9084                            <a class="w3-text" href="pandey24_interspeech.html">
9085                                <p>
9086                                    All Neural Low-latency Directional Speech Extraction
9087                                    <br>
9088                                    <span class="w3-text w3-text-theme">
9089                                        Ashutosh Pandey, Sanha Lee, Juan Azcarreta, Daniel Wong, Buye Xu
9090                                    </span>
9091                                </p>
9092                            </a>
9093                            <a class="w3-text" href="heo24_interspeech.html">
9094                                <p>
9095                                    Centroid Estimation with Transformer-Based Speaker Embedder for Robust Target Speaker Extraction
9096                                    <br>
9097                                    <span class="w3-text w3-text-theme">
9098                                        Woon-Haeng Heo, Joongyu Maeng, Yoseb Kang, Namhyun Cho
9099                                    </span>
9100                                </p>
9101                            </a>
9102                            <a class="w3-text" href="srinivas24_interspeech.html">
9103                                <p>
9104                                    Knowledge boosting during low-latency inference
9105                                    <br>
9106                                    <span class="w3-text w3-text-theme">
9107                                        Vidya Srinivas, Malek Itani, Tuochao Chen, Emre Sefik Eskimez, Takuya Yoshioka, Shyamnath Gollakota
9108                                    </span>
9109                                </p>
9110                            </a>
9111                            <a class="w3-text" href="wu24h_interspeech.html">
9112                                <p>
9113                                    Unified Audio Visual Cues for Target Speaker Extraction
9114                                    <br>
9115                                    <span class="w3-text w3-text-theme">
9116                                        Tianci Wu, Shulin He, Jiahui Pan, Haifeng Huang, Zhijian Mo, Xueliang Zhang
9117                                    </span>
9118                                </p>
9119                            </a>
9120                            <a class="w3-text" href="liu24j_interspeech.html">
9121                                <p>
9122                                    Target Speaker Extraction with Curriculum Learning
9123                                    <br>
9124                                    <span class="w3-text w3-text-theme">
9125                                        Yun Liu, Xuechen Liu, Xiaoxiao Miao, Junichi Yamagishi
9126                                    </span>
9127                                </p>
9128                            </a>
9129                        </div>
9130                    </div>
9131                    <br>
9132                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Voice Conversion 3"></div>
9133                    <div class="w3-card w3-round w3-white w3-padding">
9134                        <div class="w3-container"  style="margin-top:40px">
9135                            <h4 class="w3-center">Speech Synthesis: Voice Conversion 3</h4>
9136                            <hr>
9137                            <a class="w3-text" href="bai24_interspeech.html">
9138                                <p>
9139                                    SPA-SVC: Self-supervised Pitch Augmentation for Singing Voice Conversion
9140                                    <br>
9141                                    <span class="w3-text w3-text-theme">
9142                                        Bingsong Bai, Fengping Wang, Yingming Gao, Ya Li
9143                                    </span>
9144                                </p>
9145                            </a>
9146                            <a class="w3-text" href="salman24_interspeech.html">
9147                                <p>
9148                                    Towards Naturalistic Voice Conversion: NaturalVoices Dataset with an Automatic Processing Pipeline
9149                                    <br>
9150                                    <span class="w3-text w3-text-theme">
9151                                        Ali N. Salman, Zongyang Du, Shreeram Suresh Chandra, İsmail Rasim Ülgen, Carlos Busso, Berrak Sisman
9152                                    </span>
9153                                </p>
9154                            </a>
9155                            <a class="w3-text" href="tanaka24_interspeech.html">
9156                                <p>
9157                                    PRVAE-VC2: Non-Parallel Voice Conversion by Distillation of Speech Representations
9158                                    <br>
9159                                    <span class="w3-text w3-text-theme">
9160                                        Kou Tanaka, Hirokazu Kameoka, Takuhiro Kaneko, Yuto Kondo
9161                                    </span>
9162                                </p>
9163                            </a>
9164                            <a class="w3-text" href="niu24_interspeech.html">
9165                                <p>
9166                                    HybridVC: Efficient Voice Style Conversion with Text and Audio Prompts
9167                                    <br>
9168                                    <span class="w3-text w3-text-theme">
9169                                        Xinlei Niu, Jing Zhang, Charles Patrick Martin
9170                                    </span>
9171                                </p>
9172                            </a>
9173                            <a class="w3-text" href="hai24_interspeech.html">
9174                                <p>
9175                                    DreamVoice: Text-Guided Voice Conversion
9176                                    <br>
9177                                    <span class="w3-text w3-text-theme">
9178                                        Jiarui Hai, Karan Thakkar, Helin Wang, Zengyi Qin, Mounya Elhilali
9179                                    </span>
9180                                </p>
9181                            </a>
9182                            <a class="w3-text" href="lee24d_interspeech.html">
9183                                <p>
9184                                    Hear Your Face: Face-based voice conversion with F0 estimation
9185                                    <br>
9186                                    <span class="w3-text w3-text-theme">
9187                                        Jaejun Lee, Yoori Oh, Injune Hwang, Kyogu Lee
9188                                    </span>
9189                                </p>
9190                            </a>
9191                            <a class="w3-text" href="siriwardena24_interspeech.html">
9192                                <p>
9193                                    Accent Conversion with Articulatory Representations
9194                                    <br>
9195                                    <span class="w3-text w3-text-theme">
9196                                        Yashish M. Siriwardena, Nathan Swedlow, Audrey Howard, Evan Gitterman, Dan Darcy, Carol Espy-Wilson, Andrea Fanelli
9197                                    </span>
9198                                </p>
9199                            </a>
9200                            <a class="w3-text" href="huang24e_interspeech.html">
9201                                <p>
9202                                    USD-AC: Unsupervised Speech Disentanglement for Accent Conversion
9203                                    <br>
9204                                    <span class="w3-text w3-text-theme">
9205                                        Jen-Hung Huang, Wei-Tsung Lee, Chung-Hsien Wu
9206                                    </span>
9207                                </p>
9208                            </a>
9209                            <a class="w3-text" href="kanagawa24b_interspeech.html">
9210                                <p>
9211                                    Knowledge Distillation from Self-Supervised Representation Learning Model with Discrete Speech Units for Any-to-Any Streaming Voice Conversion
9212                                    <br>
9213                                    <span class="w3-text w3-text-theme">
9214                                        Hiroki Kanagawa, Yusuke Ijima
9215                                    </span>
9216                                </p>
9217                            </a>
9218                        </div>
9219                    </div>
9220                    <br>
9221                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Paradigms and Methods 3"></div>
9222                    <div class="w3-card w3-round w3-white w3-padding">
9223                        <div class="w3-container"  style="margin-top:40px">
9224                            <h4 class="w3-center">Speech Synthesis: Paradigms and Methods 3</h4>
9225                            <hr>
9226                            <a class="w3-text" href="yang24l_interspeech.html">
9227                                <p>
9228                                    SimpleSpeech: Towards Simple and Efficient Text-to-Speech with Scalar Latent Transformer Diffusion Models
9229                                    <br>
9230                                    <span class="w3-text w3-text-theme">
9231                                        Dongchao Yang, Dingdong Wang, Haohan Guo, Xueyuan Chen, Xixin Wu, Helen Meng
9232                                    </span>
9233                                </p>
9234                            </a>
9235                            <a class="w3-text" href="lovelace24_interspeech.html">
9236                                <p>
9237                                    Sample-Efficient Diffusion for Text-To-Speech Synthesis
9238                                    <br>
9239                                    <span class="w3-text w3-text-theme">
9240                                        Justin Lovelace, Soham Ray, Kwangyoun Kim, Kilian Q. Weinberger, Felix Wu
9241                                    </span>
9242                                </p>
9243                            </a>
9244                            <a class="w3-text" href="feng24d_interspeech.html">
9245                                <p>
9246                                    Exploring the Robustness of Text-to-Speech Synthesis Based on Diffusion Probabilistic Models to Heavily Noisy Transcriptions
9247                                    <br>
9248                                    <span class="w3-text w3-text-theme">
9249                                        Jingyi Feng, Yusuke Yasuda, Tomoki Toda
9250                                    </span>
9251                                </p>
9252                            </a>
9253                            <a class="w3-text" href="kim24_interspeech.html">
9254                                <p>
9255                                    VoiceTailor: Lightweight Plug-In Adapter for Diffusion-Based Personalized Text-to-Speech
9256                                    <br>
9257                                    <span class="w3-text w3-text-theme">
9258                                        Heeseung Kim, Sang-gil Lee, Jiheum Yeom, Che Hyun Lee, Sungwon Kim, Sungroh Yoon
9259                                    </span>
9260                                </p>
9261                            </a>
9262                            <a class="w3-text" href="sadekova24_interspeech.html">
9263                                <p>
9264                                    PitchFlow: adding pitch control to a Flow-matching based TTS model
9265                                    <br>
9266                                    <span class="w3-text w3-text-theme">
9267                                        Tasnima Sadekova, Mikhail Kudinov, Vadim Popov, Assel Yermekova, Artem Khrapov
9268                                    </span>
9269                                </p>
9270                            </a>
9271                            <a class="w3-text" href="yang24q_interspeech.html">
9272                                <p>
9273                                    DualSpeech: Enhancing Speaker-Fidelity and Text-Intelligibility Through Dual Classifier-Free Guidance
9274                                    <br>
9275                                    <span class="w3-text w3-text-theme">
9276                                        Jinhyeok Yang, Junhyeok Lee, Hyeong-Seok Choi, Seunghoon Ji, Hyeongju Kim, Juheon Lee
9277                                    </span>
9278                                </p>
9279                            </a>
9280                            <a class="w3-text" href="chen24q_interspeech.html">
9281                                <p>
9282                                    Generating Speakers by Prompting Listener Impressions for Pre-trained Multi-Speaker Text-to-Speech Systems
9283                                    <br>
9284                                    <span class="w3-text w3-text-theme">
9285                                        Zhengyang Chen, Xuechen Liu, Erica Cooper, Junichi Yamagishi, Yanmin Qian
9286                                    </span>
9287                                </p>
9288                            </a>
9289                            <a class="w3-text" href="song24b_interspeech.html">
9290                                <p>
9291                                    TacoLM: GaTed Attention Equipped Codec Language Model are Efficient Zero-Shot Text to Speech Synthesizers
9292                                    <br>
9293                                    <span class="w3-text w3-text-theme">
9294                                        Yakun Song, Zhuo Chen, Xiaofei Wang, Ziyang Ma, Guanrou Yang, Xie Chen
9295                                    </span>
9296                                </p>
9297                            </a>
9298                        </div>
9299                    </div>
9300                    <br>
9301                    <div class="w3-content" style="height:10px"  id="Privacy and Security in Speech Communication 2"></div>
9302                    <div class="w3-card w3-round w3-white w3-padding">
9303                        <div class="w3-container"  style="margin-top:40px">
9304                            <h4 class="w3-center">Privacy and Security in Speech Communication 2</h4>
9305                            <hr>
9306                            <a class="w3-text" href="ghosh24_interspeech.html">
9307                                <p>
9308                                    Anonymising Elderly and Pathological Speech: Voice Conversion Using DDSP and Query-by-Example
9309                                    <br>
9310                                    <span class="w3-text w3-text-theme">
9311                                        Suhita Ghosh, Melanie Jouaiti, Arnab Das, Yamini Sinha, Tim Polzehl, Ingo Siegert, Sebastian Stober
9312                                    </span>
9313                                </p>
9314                            </a>
9315                            <a class="w3-text" href="wang24ha_interspeech.html">
9316                                <p>
9317                                    Asynchronous Voice Anonymization Using Adversarial Perturbation On Speaker Embedding
9318                                    <br>
9319                                    <span class="w3-text w3-text-theme">
9320                                        Rui Wang, Liping Chen, Kong Aik Lee, Zhen-Hua Ling
9321                                    </span>
9322                                </p>
9323                            </a>
9324                            <a class="w3-text" href="meyer24_interspeech.html">
9325                                <p>
9326                                    Probing the Feasibility of Multilingual Speaker Anonymization
9327                                    <br>
9328                                    <span class="w3-text w3-text-theme">
9329                                        Sarina Meyer, Florian Lux, Ngoc Thang Vu
9330                                    </span>
9331                                </p>
9332                            </a>
9333                            <a class="w3-text" href="huang24_interspeech.html">
9334                                <p>
9335                                    DiffVC+: Improving Diffusion-based Voice Conversion for Speaker Anonymization
9336                                    <br>
9337                                    <span class="w3-text w3-text-theme">
9338                                        Fan Huang, Kun Zeng, Wei Zhu
9339                                    </span>
9340                                </p>
9341                            </a>
9342                        </div>
9343                    </div>
9344                    <br>
9345                    <div class="w3-content" style="height:10px"  id="Streaming ASR"></div>
9346                    <div class="w3-card w3-round w3-white w3-padding">
9347                        <div class="w3-container"  style="margin-top:40px">
9348                            <h4 class="w3-center">Streaming ASR</h4>
9349                            <hr>
9350                            <a class="w3-text" href="yang24m_interspeech.html">
9351                                <p>
9352                                    Learning from Back Chunks: Acquiring More Future Knowledge for Streaming ASR Models via Self Distillation
9353                                    <br>
9354                                    <span class="w3-text w3-text-theme">
9355                                        Yuting Yang, Guodong Ma, Yuke Li, Binbin Du, Haoqi Zhu, Liang Ruan
9356                                    </span>
9357                                </p>
9358                            </a>
9359                            <a class="w3-text" href="tsunoo24_interspeech.html">
9360                                <p>
9361                                    Decoder-only Architecture for Streaming End-to-end Speech Recognition
9362                                    <br>
9363                                    <span class="w3-text w3-text-theme">
9364                                        Emiru Tsunoo, Hayato Futami, Yosuke Kashiwagi, Siddhant Arora, Shinji Watanabe
9365                                    </span>
9366                                </p>
9367                            </a>
9368                            <a class="w3-text" href="chen24u_interspeech.html">
9369                                <p>
9370                                    Streaming Decoder-Only Automatic Speech Recognition with Discrete Speech Units: A Pilot Study
9371                                    <br>
9372                                    <span class="w3-text w3-text-theme">
9373                                        Peikun Chen, Sining Sun, Changhao Shan, Qing Yang, Lei Xie
9374                                    </span>
9375                                </p>
9376                            </a>
9377                            <a class="w3-text" href="heitkaemper24_interspeech.html">
9378                                <p>
9379                                    TfCleanformer: A streaming, array-agnostic, full- and sub-band modeling front-end for robust ASR
9380                                    <br>
9381                                    <span class="w3-text w3-text-theme">
9382                                        Jens Heitkaemper, Joe Caroselli, Arun Narayanan, Nathan Howard
9383                                    </span>
9384                                </p>
9385                            </a>
9386                            <a class="w3-text" href="le24_interspeech.html">
9387                                <p>
9388                                    Improving Streaming Speech Recognition With Time-Shifted Contextual Attention And Dynamic Right Context Masking
9389                                    <br>
9390                                    <span class="w3-text w3-text-theme">
9391                                        Khanh Le, Duc Chau
9392                                    </span>
9393                                </p>
9394                            </a>
9395                            <a class="w3-text" href="wang24ea_interspeech.html">
9396                                <p>
9397                                    Simul-Whisper: Attention-Guided Streaming Whisper with Truncation Detection
9398                                    <br>
9399                                    <span class="w3-text w3-text-theme">
9400                                        Haoyu Wang, Guoqiang Hu, Guodong Lin, Wei-Qiang Zhang, Jian Li
9401                                    </span>
9402                                </p>
9403                            </a>
9404                        </div>
9405                    </div>
9406                    <br>
9407                    <div class="w3-content" style="height:10px"  id="Computational Resource Constrained ASR"></div>
9408                    <div class="w3-card w3-round w3-white w3-padding">
9409                        <div class="w3-container"  style="margin-top:40px">
9410                            <h4 class="w3-center">Computational Resource Constrained ASR</h4>
9411                            <hr>
9412                            <a class="w3-text" href="xiao24b_interspeech.html">
9413                                <p>
9414                                    Dynamic Data Pruning for Automatic Speech Recognition
9415                                    <br>
9416                                    <span class="w3-text w3-text-theme">
9417                                        Qiao Xiao, Pingchuan Ma, Adriana Fernandez-Lopez, Boqian Wu, Lu Yin, Stavros Petridis, Mykola Pechenizkiy, Maja Pantic, Decebal Constantin Mocanu, Shiwei Liu
9418                                    </span>
9419                                </p>
9420                            </a>
9421                            <a class="w3-text" href="kim24k_interspeech.html">
9422                                <p>
9423                                    Mitigating Overfitting in Structured Pruning of ASR Models with Gradient-Guided Parameter Regularization
9424                                    <br>
9425                                    <span class="w3-text w3-text-theme">
9426                                        Dong-Hyun Kim, Joon-Hyuk Chang
9427                                    </span>
9428                                </p>
9429                            </a>
9430                            <a class="w3-text" href="gu24_interspeech.html">
9431                                <p>
9432                                    SparseWAV: Fast and Accurate One-Shot Unstructured Pruning for Large Speech Foundation Models
9433                                    <br>
9434                                    <span class="w3-text w3-text-theme">
9435                                        Tianteng Gu, Bei Liu, Hang Shao, Yanmin Qian
9436                                    </span>
9437                                </p>
9438                            </a>
9439                            <a class="w3-text" href="li24o_interspeech.html">
9440                                <p>
9441                                    One-pass Multiple Conformer and Foundation Speech Systems Compression and Quantization Using An All-in-one Neural Model
9442                                    <br>
9443                                    <span class="w3-text w3-text-theme">
9444                                        Zhaoqing Li, Haoning Xu, Tianzi Wang, Shoukang Hu, Zengrui Jin, Shujie Hu, Jiajun Deng, Mingyu Cui, Mengzhe Geng, Xunying Liu
9445                                    </span>
9446                                </p>
9447                            </a>
9448                            <a class="w3-text" href="rybakov24_interspeech.html">
9449                                <p>
9450                                    USM RNN-T model weights binarization
9451                                    <br>
9452                                    <span class="w3-text w3-text-theme">
9453                                        Oleg Rybakov, Dmitriy Serdyuk, Chengjian Zheng
9454                                    </span>
9455                                </p>
9456                            </a>
9457                            <a class="w3-text" href="lin24d_interspeech.html">
9458                                <p>
9459                                    DAISY: Data Adaptive Self-Supervised Early Exit for Speech Representation Models
9460                                    <br>
9461                                    <span class="w3-text w3-text-theme">
9462                                        Tzu-Quan Lin, Hung-yi Lee, Hao Tang
9463                                    </span>
9464                                </p>
9465                            </a>
9466                            <a class="w3-text" href="park24_interspeech.html">
9467                                <p>
9468                                    RepTor: Re-parameterizable Temporal Convolution for Keyword Spotting via Differentiable Kernel Search
9469                                    <br>
9470                                    <span class="w3-text w3-text-theme">
9471                                        Eunik Park, Daehyun Ahn, Hyungjun Kim
9472                                    </span>
9473                                </p>
9474                            </a>
9475                            <a class="w3-text" href="wang24p_interspeech.html">
9476                                <p>
9477                                    Global-Local Convolution with Spiking Neural Networks for Energy-efficient Keyword Spotting
9478                                    <br>
9479                                    <span class="w3-text w3-text-theme">
9480                                        Shuai Wang, Dehao Zhang, Kexin Shi, Yuchen Wang, Wenjie Wei, Jibin Wu, Malu Zhang
9481                                    </span>
9482                                </p>
9483                            </a>
9484                            <a class="w3-text" href="song24c_interspeech.html">
9485                                <p>
9486                                    ED-sKWS: Early-Decision Spiking Neural Networks for Rapid, and Energy-Efficient Keyword Spotting
9487                                    <br>
9488                                    <span class="w3-text w3-text-theme">
9489                                        Zeyang Song, Qianhui Liu, Qu Yang, Yizhou Peng, Haizhou Li
9490                                    </span>
9491                                </p>
9492                            </a>
9493                            <a class="w3-text" href="ling24_interspeech.html">
9494                                <p>
9495                                    A Small and Fast BERT for Chinese Medical Punctuation Restoration
9496                                    <br>
9497                                    <span class="w3-text w3-text-theme">
9498                                        Tongtao Ling, Yutao Lai, Lei Chen, Shilei Huang, Yi Liu
9499                                    </span>
9500                                </p>
9501                            </a>
9502                        </div>
9503                    </div>
9504                    <br>
9505                    <div class="w3-content" style="height:10px"  id="Evaluation of Speech Technology Systems"></div>
9506                    <div class="w3-card w3-round w3-white w3-padding">
9507                        <div class="w3-container"  style="margin-top:40px">
9508                            <h4 class="w3-center">Evaluation of Speech Technology Systems</h4>
9509                            <hr>
9510                            <a class="w3-text" href="heuser24_interspeech.html">
9511                                <p>
9512                                    Quantification of stylistic differences in human- and ASR-produced transcripts of African American English
9513                                    <br>
9514                                    <span class="w3-text w3-text-theme">
9515                                        Annika Heuser, Tyler Kendall, Miguel del Rio, Quinn McNamara, Nishchal Bhandari, Corey Miller, Migüel Jetté
9516                                    </span>
9517                                </p>
9518                            </a>
9519                            <a class="w3-text" href="kuhn24_interspeech.html">
9520                                <p>
9521                                    Beyond Levenshtein: Leveraging Multiple Algorithms for Robust Word Error Rate Computations And Granular Error Classifications
9522                                    <br>
9523                                    <span class="w3-text w3-text-theme">
9524                                        Korbinian Kuhn, Verena Kersken, Gottfried Zimmermann
9525                                    </span>
9526                                </p>
9527                            </a>
9528                            <a class="w3-text" href="teleki24_interspeech.html">
9529                                <p>
9530                                    Comparing ASR Systems in the Context of Speech Disfluencies
9531                                    <br>
9532                                    <span class="w3-text w3-text-theme">
9533                                        Maria Teleki, Xiangjue Dong, Soohwan Kim, James Caverlee
9534                                    </span>
9535                                </p>
9536                            </a>
9537                            <a class="w3-text" href="lu24d_interspeech.html">
9538                                <p>
9539                                    Deep Prosodic Features in Tandem with Perceptual Judgments of Word Reduction for Tone Recognition in Conversed Speech
9540                                    <br>
9541                                    <span class="w3-text w3-text-theme">
9542                                        Xiang-Li Lu, Yi-Fen Liu
9543                                    </span>
9544                                </p>
9545                            </a>
9546                            <a class="w3-text" href="sasindran24_interspeech.html">
9547                                <p>
9548                                    SeMaScore: A new evaluation metric for automatic speech recognition tasks
9549                                    <br>
9550                                    <span class="w3-text w3-text-theme">
9551                                        Zitha Sasindran, Harsha Yelchuri, T. V. Prabhakar
9552                                    </span>
9553                                </p>
9554                            </a>
9555                        </div>
9556                    </div>
9557                    <br>
9558                    <div class="w3-content" style="height:10px"  id="Neural Network Training for Speech Recognition"></div>
9559                    <div class="w3-card w3-round w3-white w3-padding">
9560                        <div class="w3-container"  style="margin-top:40px">
9561                            <h4 class="w3-center">Neural Network Training for Speech Recognition</h4>
9562                            <hr>
9563                            <a class="w3-text" href="xu24_interspeech.html">
9564                                <p>
9565                                    Dynamic Encoder Size Based on Data-Driven Layer-wise Pruning for Speech Recognition
9566                                    <br>
9567                                    <span class="w3-text w3-text-theme">
9568                                        Jingjing Xu, Wei Zhou, Zijian Yang, Eugen Beck, Ralf Schlüter
9569                                    </span>
9570                                </p>
9571                            </a>
9572                            <a class="w3-text" href="hou24_interspeech.html">
9573                                <p>
9574                                    Revisiting Convolution-free Transformer for Speech Recognition
9575                                    <br>
9576                                    <span class="w3-text w3-text-theme">
9577                                        Zejiang Hou, Goeric Huybrechts, Anshu Bhatia, Daniel Garcia-Romero, Kyu J. Han, Katrin Kirchhoff
9578                                    </span>
9579                                </p>
9580                            </a>
9581                            <a class="w3-text" href="huang24c_interspeech.html">
9582                                <p>
9583                                    Optimizing Large-Scale Context Retrieval for End-to-End ASR
9584                                    <br>
9585                                    <span class="w3-text w3-text-theme">
9586                                        Zhiqi Huang, Diamantino Caseiro, Kandarp Joshi, Christopher Li, Pat Rondon, Zelin Wu, Petr Zadrazil, Lillian Zhou
9587                                    </span>
9588                                </p>
9589                            </a>
9590                            <a class="w3-text" href="choi24b_interspeech.html">
9591                                <p>
9592                                    Self-Supervised Speech Representations are More Phonetic than Semantic
9593                                    <br>
9594                                    <span class="w3-text w3-text-theme">
9595                                        Kwanghee Choi, Ankita Pasad, Tomohiko Nakamura, Satoru Fukayama, Karen Livescu, Shinji Watanabe
9596                                    </span>
9597                                </p>
9598                            </a>
9599                            <a class="w3-text" href="han24_interspeech.html">
9600                                <p>
9601                                    Enhancing CTC-based speech recognition with diverse modeling units
9602                                    <br>
9603                                    <span class="w3-text w3-text-theme">
9604                                        Shiyi Han, Mingbin Xu, Zhihong Lei, Zhen Huang, Xingyu Na
9605                                    </span>
9606                                </p>
9607                            </a>
9608                            <a class="w3-text" href="kim24d_interspeech.html">
9609                                <p>
9610                                    Guiding Frame-Level CTC Alignments Using Self-knowledge Distillation
9611                                    <br>
9612                                    <span class="w3-text w3-text-theme">
9613                                        Eungbeom Kim, Hantae Kim, Kyogu Lee
9614                                    </span>
9615                                </p>
9616                            </a>
9617                        </div>
9618                    </div>
9619                    <br>
9620                    <div class="w3-content" style="height:10px"  id="Leveraging Large Language Models and Contextual Features for Phonetic Analysis (Special Session)"></div>
9621                    <div class="w3-card w3-round w3-white w3-padding">
9622                        <div class="w3-container"  style="margin-top:40px">
9623                            <h4 class="w3-center">Leveraging Large Language Models and Contextual Features for Phonetic Analysis (Special Session)</h4>
9624                            <hr>
9625                            <a class="w3-text" href="deheerkloots24_interspeech.html">
9626                                <p>
9627                                    Human-like Linguistic Biases in Neural Speech Models: Phonetic Categorization and Phonotactic Constraints in Wav2Vec2.0
9628                                    <br>
9629                                    <span class="w3-text w3-text-theme">
9630                                        Marianne de Heer Kloots, Willem Zuidema
9631                                    </span>
9632                                </p>
9633                            </a>
9634                            <a class="w3-text" href="lin24e_interspeech.html">
9635                                <p>
9636                                    Exploring Pre-trained Speech Model for Articulatory Feature Extraction in Dysarthric Speech Using ASR
9637                                    <br>
9638                                    <span class="w3-text w3-text-theme">
9639                                        Yuqin Lin, Longbiao Wang, Jianwu Dang, Nobuaki Minematsu
9640                                    </span>
9641                                </p>
9642                            </a>
9643                            <a class="w3-text" href="hao24c_interspeech.html">
9644                                <p>
9645                                    Exploring Self-Supervised Speech Representations for Cross-lingual Acoustic-to-Articulatory Inversion
9646                                    <br>
9647                                    <span class="w3-text w3-text-theme">
9648                                        Yun Hao, Reihaneh Amooie, Wietse de Vries, Thomas Tienkamp, Rik van Noord, Martijn Wieling
9649                                    </span>
9650                                </p>
9651                            </a>
9652                            <a class="w3-text" href="shams24_interspeech.html">
9653                                <p>
9654                                    Are Articulatory Feature Overlaps Shrouded in Speech Embeddings?
9655                                    <br>
9656                                    <span class="w3-text w3-text-theme">
9657                                        Erfan A. Shams, Iona Gessinger, Patrick Cormac English, Julie Carson-Berndsen
9658                                    </span>
9659                                </p>
9660                            </a>
9661                            <a class="w3-text" href="english24_interspeech.html">
9662                                <p>
9663                                    Searching for Structure: Appraising the Organisation of Speech Features in wav2vec 2.0 Embeddings
9664                                    <br>
9665                                    <span class="w3-text w3-text-theme">
9666                                        Patrick Cormac English, John D. Kelleher, Julie Carson-Berndsen
9667                                    </span>
9668                                </p>
9669                            </a>
9670                        </div>
9671                    </div>
9672                    <br>
9673                    <div class="w3-content" style="height:10px"  id="Responsible Speech Foundation Models (Special Session)"></div>
9674                    <div class="w3-card w3-round w3-white w3-padding">
9675                        <div class="w3-container"  style="margin-top:40px">
9676                            <h4 class="w3-center">Responsible Speech Foundation Models (Special Session)</h4>
9677                            <hr>
9678                            <a class="w3-text" href="wiepert24_interspeech.html">
9679                                <p>
9680                                    Speech foundation models in healthcare: Effect of layer selection on pathological speech feature prediction
9681                                    <br>
9682                                    <span class="w3-text w3-text-theme">
9683                                        Daniela A. Wiepert, Rene L. Utianski, Joseph R. Duffy, John L. Stricker, Leland R. Barnard, David T. Jones, Hugo Botha
9684                                    </span>
9685                                </p>
9686                            </a>
9687                            <a class="w3-text" href="wagner24_interspeech.html">
9688                                <p>
9689                                    Outlier Reduction with Gated Attention for Improved Post-training Quantization in Large Sequence-to-sequence Speech Foundation Models
9690                                    <br>
9691                                    <span class="w3-text w3-text-theme">
9692                                        Dominik Wagner, Ilja Baumann, Korbinian Riedhammer, Tobias Bocklet
9693                                    </span>
9694                                </p>
9695                            </a>
9696                            <a class="w3-text" href="kulkarni24_interspeech.html">
9697                                <p>
9698                                    Unveiling Biases while Embracing Sustainability: Assessing the Dual Challenges of Automatic Speech Recognition Systems
9699                                    <br>
9700                                    <span class="w3-text w3-text-theme">
9701                                        Ajinkya Kulkarni, Atharva Kulkarni, Miguel Couceiro, Isabel Trancoso
9702                                    </span>
9703                                </p>
9704                            </a>
9705                            <a class="w3-text" href="lin24i_interspeech.html">
9706                                <p>
9707                                    Emo-bias: A Large Scale Evaluation of Social Bias on Speech Emotion Recognition
9708                                    <br>
9709                                    <span class="w3-text w3-text-theme">
9710                                        Yi-Cheng Lin, Haibin Wu, Huang-Cheng Chou, Chi-Chun Lee, Hung-yi Lee
9711                                    </span>
9712                                </p>
9713                            </a>
9714                            <a class="w3-text" href="lin24b_interspeech.html">
9715                                <p>
9716                                    On the social bias of speech self-supervised models
9717                                    <br>
9718                                    <span class="w3-text w3-text-theme">
9719                                        Yi-Cheng Lin, Tzu-Quan Lin, Hsi-Che Lin, Andy T. Liu, Hung-yi Lee
9720                                    </span>
9721                                </p>
9722                            </a>
9723                            <a class="w3-text" href="chang24d_interspeech.html">
9724                                <p>
9725                                    Self-supervised Speech Representations Still Struggle with African American Vernacular English
9726                                    <br>
9727                                    <span class="w3-text w3-text-theme">
9728                                        Kalvin Chang, Yi-Hui Chou, Jiatong Shi, Hsuan-Ming Chen, Nicole Holliday, Odette Scharenborg, David R. Mortensen
9729                                    </span>
9730                                </p>
9731                            </a>
9732                            <a class="w3-text" href="aldeneh24_interspeech.html">
9733                                <p>
9734                                    Can you Remove the Downstream Model for Speaker Recognition with Self-Supervised Speech Features?
9735                                    <br>
9736                                    <span class="w3-text w3-text-theme">
9737                                        Zakaria Aldeneh, Takuya Higuchi, Jee-weon Jung, Skyler Seto, Tatiana Likhomanenko, Stephen Shum, Ahmed Hussen Abdelaziz, Shinji Watanabe, Barry-John Theobald
9738                                    </span>
9739                                </p>
9740                            </a>
9741                            <a class="w3-text" href="meng24c_interspeech.html">
9742                                <p>
9743                                    Empowering Whisper as a Joint Multi-Talker and Target-Talker Speech Recognition System
9744                                    <br>
9745                                    <span class="w3-text w3-text-theme">
9746                                        Lingwei Meng, Jiawen Kang, Yuejiao Wang, Zengrui Jin, Xixin Wu, Xunying Liu, Helen Meng
9747                                    </span>
9748                                </p>
9749                            </a>
9750                        </div>
9751                    </div>
9752                    <br>
9753                    <div class="w3-content" style="height:10px"  id="Multimodal Paralinguistics"></div>
9754                    <div class="w3-card w3-round w3-white w3-padding">
9755                        <div class="w3-container"  style="margin-top:40px">
9756                            <h4 class="w3-center">Multimodal Paralinguistics</h4>
9757                            <hr>
9758                            <a class="w3-text" href="cai24b_interspeech.html">
9759                                <p>
9760                                    LoRA-MER: Low-Rank Adaptation of Pre-Trained Speech Models for Multimodal Emotion Recognition Using Mutual Information
9761                                    <br>
9762                                    <span class="w3-text w3-text-theme">
9763                                        Yunrui Cai, Zhiyong Wu, Jia Jia, Helen Meng
9764                                    </span>
9765                                </p>
9766                            </a>
9767                            <a class="w3-text" href="li24z_interspeech.html">
9768                                <p>
9769                                    Enhancing Modal Fusion by Alignment and Label Matching for Multimodal Emotion Recognition
9770                                    <br>
9771                                    <span class="w3-text w3-text-theme">
9772                                        Qifei Li, Yingming Gao, Yuhua Wen, Cong Wang, Ya Li
9773                                    </span>
9774                                </p>
9775                            </a>
9776                            <a class="w3-text" href="zhu24_interspeech.html">
9777                                <p>
9778                                    Prompt Link Multimodal Fusion in Multimodal Sentiment Analysis
9779                                    <br>
9780                                    <span class="w3-text w3-text-theme">
9781                                        Kang Zhu, Cunhang Fan, Jianhua Tao, Zhao Lv
9782                                    </span>
9783                                </p>
9784                            </a>
9785                            <a class="w3-text" href="wang24r_interspeech.html">
9786                                <p>
9787                                    A multimodal analysis of different types of laughter expression in conversational dialogues
9788                                    <br>
9789                                    <span class="w3-text w3-text-theme">
9790                                        Kexin Wang, Carlos Ishi, Ryoko Hayashi
9791                                    </span>
9792                                </p>
9793                            </a>
9794                            <a class="w3-text" href="chochlakis24_interspeech.html">
9795                                <p>
9796                                    Tackling Missing Modalities in Audio-Visual Representation Learning Using Masked Autoencoders
9797                                    <br>
9798                                    <span class="w3-text w3-text-theme">
9799                                        Georgios Chochlakis, Chandrashekhar Lavania, Prashant Mathur, Kyu J. Han
9800                                    </span>
9801                                </p>
9802                            </a>
9803                            <a class="w3-text" href="kyung24_interspeech.html">
9804                                <p>
9805                                    Enhancing Multimodal Emotion Recognition through ASR Error Compensation and LLM Fine-Tuning
9806                                    <br>
9807                                    <span class="w3-text w3-text-theme">
9808                                        Jehyun Kyung, Serin Heo, Joon-Hyuk Chang
9809                                    </span>
9810                                </p>
9811                            </a>
9812                            <a class="w3-text" href="goncalves24_interspeech.html">
9813                                <p>
9814                                    Bridging Emotions Across Languages: Low Rank Adaptation for Multilingual Speech Emotion Recognition
9815                                    <br>
9816                                    <span class="w3-text w3-text-theme">
9817                                        Lucas Goncalves, Donita Robinson, Elizabeth Richerson, Carlos Busso
9818                                    </span>
9819                                </p>
9820                            </a>
9821                        </div>
9822                    </div>
9823                    <br>
9824                    <div class="w3-content" style="height:10px"  id="Automatic Emotion Recognition"></div>
9825                    <div class="w3-card w3-round w3-white w3-padding">
9826                        <div class="w3-container"  style="margin-top:40px">
9827                            <h4 class="w3-center">Automatic Emotion Recognition</h4>
9828                            <hr>
9829                            <a class="w3-text" href="upadhyay24_interspeech.html">
9830                                <p>
9831                                    A Layer-Anchoring Strategy for Enhancing Cross-Lingual Speech Emotion Recognition
9832                                    <br>
9833                                    <span class="w3-text w3-text-theme">
9834                                        Shreya G. Upadhyay, Carlos Busso, Chi-Chun Lee
9835                                    </span>
9836                                </p>
9837                            </a>
9838                            <a class="w3-text" href="phukan24b_interspeech.html">
9839                                <p>
9840                                    Are Paralinguistic Representations all that is needed for Speech Emotion Recognition?
9841                                    <br>
9842                                    <span class="w3-text w3-text-theme">
9843                                        Orchid Chetia Phukan, Gautam Siddharth Kashyap, Arun Balaji Buduru, Rajesh Sharma
9844                                    </span>
9845                                </p>
9846                            </a>
9847                            <a class="w3-text" href="sun24b_interspeech.html">
9848                                <p>
9849                                    MFSN: Multi-perspective Fusion Search Network For Pre-training Knowledge in Speech Emotion Recognition
9850                                    <br>
9851                                    <span class="w3-text w3-text-theme">
9852                                        Haiyang Sun, Fulin Zhang, Yingying Gao, Shilei Zhang, Zheng Lian, Junlan Feng
9853                                    </span>
9854                                </p>
9855                            </a>
9856                            <a class="w3-text" href="khaertdinov24_interspeech.html">
9857                                <p>
9858                                    Exploring Self-Supervised Multi-view Contrastive Learning for Speech Emotion Recognition with Limited Annotations
9859                                    <br>
9860                                    <span class="w3-text w3-text-theme">
9861                                        Bulat Khaertdinov, Pedro Jeruis, Annanda Sousa, Enrique Hortal
9862                                    </span>
9863                                </p>
9864                            </a>
9865                        </div>
9866                    </div>
9867                    <br>
9868                    <div class="w3-content" style="height:10px"  id="Self and Weakly-Labelled Speaker Verification"></div>
9869                    <div class="w3-card w3-round w3-white w3-padding">
9870                        <div class="w3-container"  style="margin-top:40px">
9871                            <h4 class="w3-center">Self and Weakly-Labelled Speaker Verification</h4>
9872                            <hr>
9873                            <a class="w3-text" href="wang24z_interspeech.html">
9874                                <p>
9875                                    Self-Supervised Speaker Verification with Mini-Batch Prediction Correction
9876                                    <br>
9877                                    <span class="w3-text w3-text-theme">
9878                                        Junxu Wang, Zhihua Fang, Liang He
9879                                    </span>
9880                                </p>
9881                            </a>
9882                            <a class="w3-text" href="li24q_interspeech.html">
9883                                <p>
9884                                    SCDNet: Self-supervised Learning Feature based Speaker Change Detection
9885                                    <br>
9886                                    <span class="w3-text w3-text-theme">
9887                                        Yue Li, Xinsheng Wang, Li Zhang, Lei Xie
9888                                    </span>
9889                                </p>
9890                            </a>
9891                            <a class="w3-text" href="jin24c_interspeech.html">
9892                                <p>
9893                                    Self-Supervised Learning with Multi-Head Multi-Mode Knowledge Distillation for Speaker Verification
9894                                    <br>
9895                                    <span class="w3-text w3-text-theme">
9896                                        Zezhong Jin, Youzhi Tu, Man-Wai Mak
9897                                    </span>
9898                                </p>
9899                            </a>
9900                            <a class="w3-text" href="selvakumar24_interspeech.html">
9901                                <p>
9902                                    Getting More for Less: Using Weak Labels and AV-Mixup for Robust Audio-Visual Speaker Verification
9903                                    <br>
9904                                    <span class="w3-text w3-text-theme">
9905                                        Anith Selvakumar, Homa Fashandi
9906                                    </span>
9907                                </p>
9908                            </a>
9909                        </div>
9910                    </div>
9911                    <br>
9912                    <div class="w3-content" style="height:10px"  id="Acoustic Event Detection, Segmentation and Classification"></div>
9913                    <div class="w3-card w3-round w3-white w3-padding">
9914                        <div class="w3-container"  style="margin-top:40px">
9915                            <h4 class="w3-center">Acoustic Event Detection, Segmentation and Classification</h4>
9916                            <hr>
9917                            <a class="w3-text" href="behera24_interspeech.html">
9918                                <p>
9919                                    FastAST: Accelerating Audio Spectrogram Transformer via Token Merging and Cross-Model Knowledge Distillation
9920                                    <br>
9921                                    <span class="w3-text w3-text-theme">
9922                                        Swarup Ranjan Behera, Abhishek Dhiman, Karthik Gowda, Aalekhya Satya Narayani
9923                                    </span>
9924                                </p>
9925                            </a>
9926                            <a class="w3-text" href="xiao24_interspeech.html">
9927                                <p>
9928                                    LungAdapter: Efficient Adapting Audio Spectrogram Transformer for Lung Sound Classification
9929                                    <br>
9930                                    <span class="w3-text w3-text-theme">
9931                                        Li Xiao, Lucheng Fang, Yuhong Yang, Weiping Tu
9932                                    </span>
9933                                </p>
9934                            </a>
9935                            <a class="w3-text" href="feng24c_interspeech.html">
9936                                <p>
9937                                    ElasticAST: An Audio Spectrogram Transformer for All Length and Resolutions
9938                                    <br>
9939                                    <span class="w3-text w3-text-theme">
9940                                        Jiu Feng, Mehmet Hamza Erol, Joon Son Chung, Arda Senocak
9941                                    </span>
9942                                </p>
9943                            </a>
9944                            <a class="w3-text" href="omine24_interspeech.html">
9945                                <p>
9946                                    Robust Laughter Segmentation with Automatic Diverse Data Synthesis
9947                                    <br>
9948                                    <span class="w3-text w3-text-theme">
9949                                        Taisei Omine, Kenta Akita, Reiji Tsuruno
9950                                    </span>
9951                                </p>
9952                            </a>
9953                            <a class="w3-text" href="lebourdais24_interspeech.html">
9954                                <p>
9955                                    Explainable by-design Audio Segmentation through Non-Negative Matrix Factorization and Probing
9956                                    <br>
9957                                    <span class="w3-text w3-text-theme">
9958                                        Martin Lebourdais, Théo Mariotte, Antonio Almudévar, Marie Tahon, Alfonso Ortega
9959                                    </span>
9960                                </p>
9961                            </a>
9962                            <a class="w3-text" href="elbanna24_interspeech.html">
9963                                <p>
9964                                    Predicting Heart Activity from Speech using Data-driven and Knowledge-based features
9965                                    <br>
9966                                    <span class="w3-text w3-text-theme">
9967                                        Gasser Elbanna, Zohreh Mostaani, Mathew Magimai.-Doss
9968                                    </span>
9969                                </p>
9970                            </a>
9971                            <a class="w3-text" href="morozova24_interspeech.html">
9972                                <p>
9973                                    Measuring acoustic dissimilarity of hierarchical markers in task-oriented dialogue with MFCC-based dynamic time warping
9974                                    <br>
9975                                    <span class="w3-text w3-text-theme">
9976                                        Natalia Morozova, Guanghao You, Sabine Stoll, Adrian Bangerter
9977                                    </span>
9978                                </p>
9979                            </a>
9980                            <a class="w3-text" href="buddi24_interspeech.html">
9981                                <p>
9982                                    Comparative Analysis of Personalized Voice Activity Detection Systems: Assessing Real-World Effectiveness
9983                                    <br>
9984                                    <span class="w3-text w3-text-theme">
9985                                        Sai Srujana Buddi, Satyam Kumar, Utkarsh Sarawgi, Vineet Garg, Shivesh Ranjan, Ognjen Rudovic, Ahmed Hussen Abdelaziz, Saurabh Adya
9986                                    </span>
9987                                </p>
9988                            </a>
9989                            <a class="w3-text" href="wang24ca_interspeech.html">
9990                                <p>
9991                                    Generalized Fake Audio Detection via Deep Stable Learning
9992                                    <br>
9993                                    <span class="w3-text w3-text-theme">
9994                                        Zhiyong Wang, Ruibo Fu, Zhengqi Wen, Yuankun Xie, Yukun Liu, Xiaopeng Wang, Xuefei Liu, Yongwei Li, Jianhua Tao, Xin Qi, Yi Lu, Shuchen Shi
9995                                    </span>
9996                                </p>
9997                            </a>
9998                            <a class="w3-text" href="palaskar24_interspeech.html">
9999                                <p>
10000                                    Multimodal Large Language Models with Fusion Low Rank Adaptation for Device Directed Speech Detection
10001                                    <br>
10002                                    <span class="w3-text w3-text-theme">
10003                                        Shruti Palaskar, Ognjen Rudovic, Sameer Dharur, Florian Pesce, Gautam Krishna, Aswin Sivaraman, Jack Berkowitz, Ahmed Hussen Abdelaziz, Saurabh Adya, Ahmed Tewfik
10004                                    </span>
10005                                </p>
10006                            </a>
10007                            <a class="w3-text" href="zang24_interspeech.html">
10008                                <p>
10009                                    CtrSVDD: A Benchmark Dataset and Baseline Analysis for Controlled Singing Voice Deepfake Detection
10010                                    <br>
10011                                    <span class="w3-text w3-text-theme">
10012                                        Yongyi Zang, Jiatong Shi, You Zhang, Ryuichi Yamamoto, Jionghao Han, Yuxun Tang, Shengyuan Xu, Wenxiao Zhao, Jing Guo, Tomoki Toda, Zhiyao Duan
10013                                    </span>
10014                                </p>
10015                            </a>
10016                            <a class="w3-text" href="si24_interspeech.html">
10017                                <p>
10018                                    Fully Few-shot Class-incremental Audio Classification Using Expandable Dual-embedding Extractor
10019                                    <br>
10020                                    <span class="w3-text w3-text-theme">
10021                                        Yongjie Si, Yanxiong Li, Jialong Li, Jiaxin Tan, Qianhua He
10022                                    </span>
10023                                </p>
10024                            </a>
10025                            <a class="w3-text" href="a24_interspeech.html">
10026                                <p>
10027                                    Multi-label Bird Species Classification from Field Recordings using Mel_Graph-GCN Framework
10028                                    <br>
10029                                    <span class="w3-text w3-text-theme">
10030                                        Noumida A, Rajeev Rajan
10031                                    </span>
10032                                </p>
10033                            </a>
10034                        </div>
10035                    </div>
10036                    <br>
10037                    <div class="w3-content" style="height:10px"  id="Speech and Audio Modelling"></div>
10038                    <div class="w3-card w3-round w3-white w3-padding">
10039                        <div class="w3-container"  style="margin-top:40px">
10040                            <h4 class="w3-center">Speech and Audio Modelling</h4>
10041                            <hr>
10042                            <a class="w3-text" href="li24ha_interspeech.html">
10043                                <p>
10044                                    DiveSound: LLM-Assisted Automatic Taxonomy Construction for Diverse Audio Generation
10045                                    <br>
10046                                    <span class="w3-text w3-text-theme">
10047                                        Baihan Li, Zeyu Xie, Xuenan Xu, Yiwei Guo, Ming Yan, Ji Zhang, Kai Yu, Mengyue Wu
10048                                    </span>
10049                                </p>
10050                            </a>
10051                            <a class="w3-text" href="wang24d_interspeech.html">
10052                                <p>
10053                                    Leveraging Language Model Capabilities for Sound Event Detection
10054                                    <br>
10055                                    <span class="w3-text w3-text-theme">
10056                                        Hualei Wang, Jianguo Mao, Zhifang Guo, Jiarui Wan, Hong Liu, Xiangdong Wang
10057                                    </span>
10058                                </p>
10059                            </a>
10060                            <a class="w3-text" href="xu24f_interspeech.html">
10061                                <p>
10062                                    Enhancing Zero-shot Audio Classification using Sound Attribute Knowledge from Large Language Models
10063                                    <br>
10064                                    <span class="w3-text w3-text-theme">
10065                                        Xuenan Xu, Pingyue Zhang, Ming Yan, Ji Zhang, Mengyue Wu
10066                                    </span>
10067                                </p>
10068                            </a>
10069                            <a class="w3-text" href="guan24b_interspeech.html">
10070                                <p>
10071                                    LAFMA: A Latent Flow Matching Model for Text-to-Audio Generation
10072                                    <br>
10073                                    <span class="w3-text w3-text-theme">
10074                                        Wenhao Guan, Kaidi Wang, Wangjin Zhou, Yang Wang, Feng Deng, Hui Wang, Lin Li, Qingyang Hong, Yong Qin
10075                                    </span>
10076                                </p>
10077                            </a>
10078                            <a class="w3-text" href="cumlin24_interspeech.html">
10079                                <p>
10080                                    DNSMOS Pro: A Reduced-Size DNN for Probabilistic MOS of Speech
10081                                    <br>
10082                                    <span class="w3-text w3-text-theme">
10083                                        Fredrik Cumlin, Xinyu Liang, Victor Ungureanu, Chandan K. A. Reddy, Christian Schüldt, Saikat Chatterjee
10084                                    </span>
10085                                </p>
10086                            </a>
10087                            <a class="w3-text" href="boukun24_interspeech.html">
10088                                <p>
10089                                    Blind Zero-Shot Audio Restoration: A Variational Autoencoder Approach for Denoising and Inpainting
10090                                    <br>
10091                                    <span class="w3-text w3-text-theme">
10092                                        Veranika Boukun, Jakob Drefs, Jörg Lücke
10093                                    </span>
10094                                </p>
10095                            </a>
10096                        </div>
10097                    </div>
10098                    <br>
10099                    <div class="w3-content" style="height:10px"  id="Fake Audio Detection"></div>
10100                    <div class="w3-card w3-round w3-white w3-padding">
10101                        <div class="w3-container"  style="margin-top:40px">
10102                            <h4 class="w3-center">Fake Audio Detection</h4>
10103                            <hr>
10104                            <a class="w3-text" href="pascu24_interspeech.html">
10105                                <p>
10106                                    Towards generalisable and calibrated audio deepfake detection with self-supervised representations
10107                                    <br>
10108                                    <span class="w3-text w3-text-theme">
10109                                        Octavian Pascu, Adriana Stan, Dan Oneata, Elisabeta Oneata, Horia Cucu
10110                                    </span>
10111                                </p>
10112                            </a>
10113                            <a class="w3-text" href="xie24_interspeech.html">
10114                                <p>
10115                                    Generalized Source Tracing: Detecting Novel Audio Deepfake Algorithm with Real Emphasis and Fake Dispersion Strategy
10116                                    <br>
10117                                    <span class="w3-text w3-text-theme">
10118                                        Yuankun Xie, Ruibo Fu, Zhengqi Wen, Zhiyong Wang, Xiaopeng Wang, Haonnan Cheng, Long Ye, Jianhua Tao
10119                                    </span>
10120                                </p>
10121                            </a>
10122                            <a class="w3-text" href="zhong24_interspeech.html">
10123                                <p>
10124                                    Enhancing Partially Spoofed Audio Localization with Boundary-aware Attention Mechanism
10125                                    <br>
10126                                    <span class="w3-text w3-text-theme">
10127                                        Jiafeng Zhong, Bin Li, Jiangyan Yi
10128                                    </span>
10129                                </p>
10130                            </a>
10131                            <a class="w3-text" href="chen24o_interspeech.html">
10132                                <p>
10133                                    Singing Voice Graph Modeling for SingFake Detection
10134                                    <br>
10135                                    <span class="w3-text w3-text-theme">
10136                                        Xuanjun Chen, Haibin Wu, Roger Jang, Hung-yi Lee
10137                                    </span>
10138                                </p>
10139                            </a>
10140                            <a class="w3-text" href="wang24ga_interspeech.html">
10141                                <p>
10142                                    Genuine-Focused Learning using Mask AutoEncoder for Generalized Fake Audio Detection
10143                                    <br>
10144                                    <span class="w3-text w3-text-theme">
10145                                        Xiaopeng Wang, Ruibo Fu, Zhengqi Wen, Zhiyong Wang, Yuankun Xie, Yukun Liu, Jianhua Tao, Xuefei Liu, Yongwei Li, Xin Qi, Yi Lu, Shuchen Shi
10146                                    </span>
10147                                </p>
10148                            </a>
10149                            <a class="w3-text" href="kim24b_interspeech.html">
10150                                <p>
10151                                    One-class learning with adaptive centroid shift for audio deepfake detection
10152                                    <br>
10153                                    <span class="w3-text w3-text-theme">
10154                                        Hyun Myung Kim, Kangwook Jang, Hoirin Kim
10155                                    </span>
10156                                </p>
10157                            </a>
10158                        </div>
10159                    </div>
10160                    <br>
10161                    <div class="w3-content" style="height:10px"  id="Deep Learning-Based Speech Enhancement: Approaches, Scalability, and Evaluation"></div>
10162                    <div class="w3-card w3-round w3-white w3-padding">
10163                        <div class="w3-container"  style="margin-top:40px">
10164                            <h4 class="w3-center">Deep Learning-Based Speech Enhancement: Approaches, Scalability, and Evaluation</h4>
10165                            <hr>
10166                            <a class="w3-text" href="cao24_interspeech.html">
10167                                <p>
10168                                    VoiCor: A Residual Iterative Voice Correction Framework for Monaural Speech Enhancement
10169                                    <br>
10170                                    <span class="w3-text w3-text-theme">
10171                                        Rui Cao, Tianrui Wang, Meng Ge, Andong Li, Longbiao Wang, Jianwu Dang, Yungang Jia
10172                                    </span>
10173                                </p>
10174                            </a>
10175                            <a class="w3-text" href="parnamaa24_interspeech.html">
10176                                <p>
10177                                    Personalized Speech Enhancement Without a Separate Speaker Embedding Model
10178                                    <br>
10179                                    <span class="w3-text w3-text-theme">
10180                                        Tanel Pärnamaa, Ando Saabas
10181                                    </span>
10182                                </p>
10183                            </a>
10184                            <a class="w3-text" href="zhang24h_interspeech.html">
10185                                <p>
10186                                    URGENT Challenge: Universality, Robustness, and Generalizability For Speech Enhancement
10187                                    <br>
10188                                    <span class="w3-text w3-text-theme">
10189                                        Wangyou Zhang, Robin Scheibler, Kohei Saijo, Samuele Cornell, Chenda Li, Zhaoheng Ni, Jan Pirklbauer, Marvin Sach, Shinji Watanabe, Tim Fingscheidt, Yanmin Qian
10190                                    </span>
10191                                </p>
10192                            </a>
10193                            <a class="w3-text" href="richter24_interspeech.html">
10194                                <p>
10195                                    EARS: An Anechoic Fullband Speech Dataset Benchmarked for Speech Enhancement and Dereverberation
10196                                    <br>
10197                                    <span class="w3-text w3-text-theme">
10198                                        Julius Richter, Yi-Chiao Wu, Steven Krenn, Simon Welker, Bunlong Lay, Shinji Watanabe, Alexander Richard, Timo Gerkmann
10199                                    </span>
10200                                </p>
10201                            </a>
10202                        </div>
10203                    </div>
10204                    <br>
10205                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Other Topics 1"></div>
10206                    <div class="w3-card w3-round w3-white w3-padding">
10207                        <div class="w3-container"  style="margin-top:40px">
10208                            <h4 class="w3-center">Speech Synthesis: Other Topics 1</h4>
10209                            <hr>
10210                            <a class="w3-text" href="tan24c_interspeech.html">
10211                                <p>
10212                                    LiteFocus: Accelerated Diffusion Inference for Long Audio Synthesis
10213                                    <br>
10214                                    <span class="w3-text w3-text-theme">
10215                                        Zhenxiong Tan, Xinyin Ma, Gongfan Fang, Xinchao Wang
10216                                    </span>
10217                                </p>
10218                            </a>
10219                            <a class="w3-text" href="kim24e_interspeech.html">
10220                                <p>
10221                                    Speak in the Scene: Diffusion-based Acoustic Scene Transfer toward Immersive Speech Generation
10222                                    <br>
10223                                    <span class="w3-text w3-text-theme">
10224                                        Miseul Kim, Soo-Whan Chung, Youna Ji, Hong-Goo Kang, Min-Seok Choi
10225                                    </span>
10226                                </p>
10227                            </a>
10228                            <a class="w3-text" href="li24y_interspeech.html">
10229                                <p>
10230                                    PL-TTS: A Generalizable Prompt-based Diffusion TTS Augmented by Large Language Model
10231                                    <br>
10232                                    <span class="w3-text w3-text-theme">
10233                                        Shuhua Li, Qirong Mao, Jiatong Shi
10234                                    </span>
10235                                </p>
10236                            </a>
10237                            <a class="w3-text" href="abel24_interspeech.html">
10238                                <p>
10239                                    Towards realtime co-speech gestures synthesis using STARGATE
10240                                    <br>
10241                                    <span class="w3-text w3-text-theme">
10242                                        Louis Abel, Vincent Colotte, Slim Ouni
10243                                    </span>
10244                                </p>
10245                            </a>
10246                            <a class="w3-text" href="shi24f_interspeech.html">
10247                                <p>
10248                                    PPPR: Portable Plug-in Prompt Refiner for Text to Audio Generation
10249                                    <br>
10250                                    <span class="w3-text w3-text-theme">
10251                                        Shuchen Shi, Ruibo Fu, Zhengqi Wen, Jianhua Tao, Tao Wang, Chunyu Qiang, Yi Lu, Xin Qi, Xuefei Liu, Yukun Liu, Yongwei Li, Zhiyong Wang, Xiaopeng Wang
10252                                    </span>
10253                                </p>
10254                            </a>
10255                            <a class="w3-text" href="lee24m_interspeech.html">
10256                                <p>
10257                                    Neural ATSM: Fully Neural Network-based Adaptive Time-Scale Modification Using Sentence-Specific Dynamic Control
10258                                    <br>
10259                                    <span class="w3-text w3-text-theme">
10260                                        Jaeuk Lee, Sohee Jang, Joon-Hyuk Chang
10261                                    </span>
10262                                </p>
10263                            </a>
10264                            <a class="w3-text" href="guo24c_interspeech.html">
10265                                <p>
10266                                    FLY-TTS: Fast, Lightweight and High-Quality End-to-End Text-to-Speech Synthesis
10267                                    <br>
10268                                    <span class="w3-text w3-text-theme">
10269                                        Yinlin Guo, Yening Lv, Jinqiao Dou, Yan Zhang, Yuehai Wang
10270                                    </span>
10271                                </p>
10272                            </a>
10273                            <a class="w3-text" href="kunesova24_interspeech.html">
10274                                <p>
10275                                    Zero-shot Out-of-domain is No Joke: Lessons Learned in the VoiceMOS 2023 MOS Prediction Challenge
10276                                    <br>
10277                                    <span class="w3-text w3-text-theme">
10278                                        Marie Kunešová, Jan Lehečka, Josef Michálek, Jindrich Matousek, Jan Švec
10279                                    </span>
10280                                </p>
10281                            </a>
10282                            <a class="w3-text" href="ward24_interspeech.html">
10283                                <p>
10284                                    Towards a General-Purpose Model of Perceived Pragmatic Similarity
10285                                    <br>
10286                                    <span class="w3-text w3-text-theme">
10287                                        Nigel G. Ward, Andres Segura, Alejandro Ceballos, Divette Marco
10288                                    </span>
10289                                </p>
10290                            </a>
10291                        </div>
10292                    </div>
10293                    <br>
10294                    <div class="w3-content" style="height:10px"  id="Speech Synthesis: Other Topics 2"></div>
10295                    <div class="w3-card w3-round w3-white w3-padding">
10296                        <div class="w3-container"  style="margin-top:40px">
10297                            <h4 class="w3-center">Speech Synthesis: Other Topics 2</h4>
10298                            <hr>
10299                            <a class="w3-text" href="ratsep24_interspeech.html">
10300                                <p>
10301                                    Enabling Conversational Speech Synthesis using Noisy Spontaneous Data
10302                                    <br>
10303                                    <span class="w3-text w3-text-theme">
10304                                        Liisa Rätsep, Rasmus Lellep, Mark Fishel
10305                                    </span>
10306                                </p>
10307                            </a>
10308                            <a class="w3-text" href="yang24c_interspeech.html">
10309                                <p>
10310                                    Frame-Wise Breath Detection with Self-Training: An Exploration of Enhancing Breath Naturalness in Text-to-Speech
10311                                    <br>
10312                                    <span class="w3-text w3-text-theme">
10313                                        Dong Yang, Tomoki Koriyama, Yuki Saito
10314                                    </span>
10315                                </p>
10316                            </a>
10317                            <a class="w3-text" href="seong24_interspeech.html">
10318                                <p>
10319                                    H4C-TTS: Leveraging Multi-Modal Historical Context for Conversational Text-to-Speech
10320                                    <br>
10321                                    <span class="w3-text w3-text-theme">
10322                                        Donghyun Seong, Joon-Hyuk Chang
10323                                    </span>
10324                                </p>
10325                            </a>
10326                            <a class="w3-text" href="yang24i_interspeech.html">
10327                                <p>
10328                                    Bilingual and Code-switching TTS Enhanced with Denoising Diffusion Model and GAN
10329                                    <br>
10330                                    <span class="w3-text w3-text-theme">
10331                                        Huai-Zhe Yang, Chia-Ping Chen, Shan-Yun He, Cheng-Ruei Li
10332                                    </span>
10333                                </p>
10334                            </a>
10335                            <a class="w3-text" href="saeki24_interspeech.html">
10336                                <p>
10337                                    SpeechBERTScore: Reference-Aware Automatic Evaluation of Speech Generation Leveraging NLP Evaluation Metrics
10338                                    <br>
10339                                    <span class="w3-text w3-text-theme">
10340                                        Takaaki Saeki, Soumi Maiti, Shinnosuke Takamichi, Shinji Watanabe, Hiroshi Saruwatari
10341                                    </span>
10342                                </p>
10343                            </a>
10344                            <a class="w3-text" href="yoon24b_interspeech.html">
10345                                <p>
10346                                    UNIQUE : Unsupervised Network for Integrated Speech Quality Evaluation
10347                                    <br>
10348                                    <span class="w3-text w3-text-theme">
10349                                        Juhwan Yoon, WooSeok Ko, Seyun Um, Sungwoong Hwang, Soojoong Hwang, Changhwan Kim, Hong-Goo Kang
10350                                    </span>
10351                                </p>
10352                            </a>
10353                            <a class="w3-text" href="lee24_interspeech.html">
10354                                <p>
10355                                    FVTTS : Face Based Voice Synthesis for Text-to-Speech
10356                                    <br>
10357                                    <span class="w3-text w3-text-theme">
10358                                        Minyoung Lee, Eunil Park, Sungeun Hong
10359                                    </span>
10360                                </p>
10361                            </a>
10362                        </div>
10363                    </div>
10364                    <br>
10365                    <div class="w3-content" style="height:10px"  id="Speech synthesis: Cross-lingual and multilingual aspects"></div>
10366                    <div class="w3-card w3-round w3-white w3-padding">
10367                        <div class="w3-container"  style="margin-top:40px">
10368                            <h4 class="w3-center">Speech synthesis: Cross-lingual and multilingual aspects</h4>
10369                            <hr>
10370                            <a class="w3-text" href="lux24_interspeech.html">
10371                                <p>
10372                                    Meta Learning Text-to-Speech Synthesis in over 7000 Languages
10373                                    <br>
10374                                    <span class="w3-text w3-text-theme">
10375                                        Florian Lux, Sarina Meyer, Lyonel Behringer, Frank Zalkow, Phat Do, Matt Coler, Emanuël A. P. Habets, Ngoc Thang Vu
10376                                    </span>
10377                                </p>
10378                            </a>
10379                            <a class="w3-text" href="gong24c_interspeech.html">
10380                                <p>
10381                                    An Initial Investigation of Language Adaptation for TTS Systems under Low-resource Scenarios
10382                                    <br>
10383                                    <span class="w3-text w3-text-theme">
10384                                        Cheng Gong, Erica Cooper, Xin Wang, Chunyu Qiang, Mengzhe Geng, Dan Wells, Longbiao Wang, Jianwu Dang, Marc Tessier, Aidan Pine, Korin Richmond, Junichi Yamagishi
10385                                    </span>
10386                                </p>
10387                            </a>
10388                            <a class="w3-text" href="wu24f_interspeech.html">
10389                                <p>
10390                                    Improving Multilingual Text-to-Speech with Mixture-of-Language-Experts and Accent Disentanglement
10391                                    <br>
10392                                    <span class="w3-text w3-text-theme">
10393                                        Jing Wu, Ting Chen, Minchuan Chen, Wei Hu, Shaojun Wang, Jing Xiao
10394                                    </span>
10395                                </p>
10396                            </a>
10397                            <a class="w3-text" href="xu24g_interspeech.html">
10398                                <p>
10399                                    Seamless Language Expansion: Enhancing Multilingual Mastery in Self-Supervised Models
10400                                    <br>
10401                                    <span class="w3-text w3-text-theme">
10402                                        Jing Xu, Minglin Wu, Xixin Wu, Helen Meng
10403                                    </span>
10404                                </p>
10405                            </a>
10406                            <a class="w3-text" href="casanova24_interspeech.html">
10407                                <p>
10408                                    XTTS: a Massively Multilingual Zero-Shot Text-to-Speech Model
10409                                    <br>
10410                                    <span class="w3-text w3-text-theme">
10411                                        Edresson Casanova, Kelly Davis, Eren Gölge, Görkem Göknar, Iulian Gulea, Logan Hart, Aya Aljafari, Joshua Meyer, Reuben Morais, Samuel Olayemi, Julian Weber
10412                                    </span>
10413                                </p>
10414                            </a>
10415                            <a class="w3-text" href="guo24b_interspeech.html">
10416                                <p>
10417                                    X-E-Speech: Joint Training Framework of Non-Autoregressive Cross-lingual Emotional Text-to-Speech and Voice Conversion
10418                                    <br>
10419                                    <span class="w3-text w3-text-theme">
10420                                        Houjian Guo, Chaoran Liu, Carlos Toshinori Ishi, Hiroshi Ishiguro
10421                                    </span>
10422                                </p>
10423                            </a>
10424                        </div>
10425                    </div>
10426                    <br>
10427                    <div class="w3-content" style="height:10px"  id="Noise, Far-Field, Multi-Talker, Enhancement, Audio Classification"></div>
10428                    <div class="w3-card w3-round w3-white w3-padding">
10429                        <div class="w3-container"  style="margin-top:40px">
10430                            <h4 class="w3-center">Noise, Far-Field, Multi-Talker, Enhancement, Audio Classification</h4>
10431                            <hr>
10432                            <a class="w3-text" href="shao24_interspeech.html">
10433                                <p>
10434                                    RIR-SF: Room Impulse Response Based Spatial Feature for Target Speech Recognition in Multi-Channel Multi-Speaker Scenarios
10435                                    <br>
10436                                    <span class="w3-text w3-text-theme">
10437                                        Yiwen Shao, Shi-Xiong Zhang, Dong Yu
10438                                    </span>
10439                                </p>
10440                            </a>
10441                            <a class="w3-text" href="shao24b_interspeech.html">
10442                                <p>
10443                                    Multi-Channel Multi-Speaker ASR Using Target Speaker’s Solo Segment
10444                                    <br>
10445                                    <span class="w3-text w3-text-theme">
10446                                        Yiwen Shao, Shi-Xiong Zhang, Yong Xu, Meng Yu, Dong Yu, Daniel Povey, Sanjeev Khudanpur
10447                                    </span>
10448                                </p>
10449                            </a>
10450                            <a class="w3-text" href="ravenscroft24_interspeech.html">
10451                                <p>
10452                                    Transcription-Free Fine-Tuning of Speech Separation Models for Noisy and Reverberant Multi-Speaker Automatic Speech Recognition
10453                                    <br>
10454                                    <span class="w3-text w3-text-theme">
10455                                        William Ravenscroft, George Close, Stefan Goetze, Thomas Hain, Mohammad Soleymanpour, Anurag Chowdhur
10455y, Mark C. Fuhs
10456                                    </span>
10457                                </p>
10458                            </a>
10459                            <a class="w3-text" href="vinnikov24_interspeech.html">
10460                                <p>
10461                                    NOTSOFAR-1 Challenge: New Datasets, Baseline, and Tasks for Distant Meeting Transcription
10462                                    <br>
10463                                    <span class="w3-text w3-text-theme">
10464                                        Alon Vinnikov, Amir Ivry, Aviv Hurvitz, Igor Abramovski, Sharon Koubi, Ilya Gurvich, Shai Peer, Xiong Xiao, Benjamin Martinez Elizalde, Naoyuki Kanda, Xiaofei Wang, Shalev Shaer, Stav Yagev, Yossi Asher, Sunit Sivasankaran, Yifan Gong, Min Tang, Huaming Wang, Eyal Krupka
10465                                    </span>
10466                                </p>
10467                            </a>
10468                            <a class="w3-text" href="dissen24_interspeech.html">
10469                                <p>
10470                                    Enhanced ASR Robustness to Packet Loss with a Front-End Adaptation Network
10471                                    <br>
10472                                    <span class="w3-text w3-text-theme">
10473                                        Yehoshua Dissen, Shiry Yonash, Israel Cohen, Joseph Keshet
10474                                    </span>
10475                                </p>
10476                            </a>
10477                            <a class="w3-text" href="haider24_interspeech.html">
10478                                <p>
10479                                    Hold Me Tight: Stable Encoder-Decoder Design for Speech Enhancement
10480                                    <br>
10481                                    <span class="w3-text w3-text-theme">
10482                                        Daniel Haider, Felix Perfler, Vincent Lostanlen, Martin Ehler, Peter Balazs
10483                                    </span>
10484                                </p>
10485                            </a>
10486                            <a class="w3-text" href="wang24da_interspeech.html">
10487                                <p>
10488                                    DGSRN: Noise-Robust Speech Recognition Method with Dual-Path Gated Spectral Refinement Network
10489                                    <br>
10490                                    <span class="w3-text w3-text-theme">
10491                                        Wenjun Wang, Shangbin Mo, Ling Dong, Zhengtao Yu, Junjun Guo, Yuxin Huang
10492                                    </span>
10493                                </p>
10494                            </a>
10495                            <a class="w3-text" href="singh24b_interspeech.html">
10496                                <p>
10497                                    Towards Robust Few-shot Class Incremental Learning in Audio Classification using Contrastive Representation
10498                                    <br>
10499                                    <span class="w3-text w3-text-theme">
10500                                        Riyansha Singh, Parinita Nema, Vinod K Kurmi
10501                                    </span>
10502                                </p>
10503                            </a>
10504                            <a class="w3-text" href="sheikh24_interspeech.html">
10505                                <p>
10506                                    Bird Whisperer: Leveraging Large Pre-trained Acoustic Model for Bird Call Classification
10507                                    <br>
10508                                    <span class="w3-text w3-text-theme">
10509                                        Muhammad Umer Sheikh, Hassan Abid, Bhuiyan Sanjid Shafique, Asif Hanif, Muhammad Haris Khan
10510                                    </span>
10511                                </p>
10512                            </a>
10513                            <a class="w3-text" href="sato24_interspeech.html">
10514                                <p>
10515                                    SpeakerBeam-SS: Real-time Target Speaker Extraction with Lightweight Conv-TasNet and State Space Modeling
10516                                    <br>
10517                                    <span class="w3-text w3-text-theme">
10518                                        Hiroshi Sato, Takafumi Moriya, Masato Mimura, Shota Horiguchi, Tsubasa Ochiai, Takanori Ashihara, Atsushi Ando, Kentaro Shinayama, Marc Delcroix
10519                                    </span>
10520                                </p>
10521                            </a>
10522                            <a class="w3-text" href="borsdorf24_interspeech.html">
10523                                <p>
10524                                    wTIMIT2mix: A Cocktail Party Mixtures Database to Study Target Speaker Extraction for Normal and Whispered Speech
10525                                    <br>
10526                                    <span class="w3-text w3-text-theme">
10527                                        Marvin Borsdorf, Zexu Pan, Haizhou Li, Tanja Schultz
10528                                    </span>
10529                                </p>
10530                            </a>
10531                        </div>
10532                    </div>
10533                    <br>
10534                    <div class="w3-content" style="height:10px"  id="Self-Supervised Learning for ASR"></div>
10535                    <div class="w3-card w3-round w3-white w3-padding">
10536                        <div class="w3-container"  style="margin-top:40px">
10537                            <h4 class="w3-center">Self-Supervised Learning for ASR</h4>
10538                            <hr>
10539                            <a class="w3-text" href="getman24_interspeech.html">
10540                                <p>
10541                                    What happens in continued pre-training? Analysis of self-supervised speech models with continued pre-training for colloquial Finnish ASR
10542                                    <br>
10543                                    <span class="w3-text w3-text-theme">
10544                                        Yaroslav Getman, Tamas Grosz, Mikko Kurimo
10545                                    </span>
10546                                </p>
10547                            </a>
10548                            <a class="w3-text" href="kato24_interspeech.html">
10549                                <p>
10550                                    Self-Supervised Learning for ASR Pre-Training with Uniquely Determined Target Labels and Controlling Cepstrum Truncation for Speech Augmentation
10551                                    <br>
10552                                    <span class="w3-text w3-text-theme">
10553                                        Akihiro Kato, Hiroyuki Nagano, Kohei Chike, Masaki Nose
10554                                    </span>
10555                                </p>
10556                            </a>
10557                            <a class="w3-text" href="yadav24b_interspeech.html">
10558                                <p>
10559                                    MS-HuBERT: Mitigating Pre-training and Inference Mismatch in Masked Language Modelling methods for learning Speech Representations
10560                                    <br>
10561                                    <span class="w3-text w3-text-theme">
10562                                        Hemant Yadav, Sunayana Sitaram, Rajiv Ratn Shah
10563                                    </span>
10564                                </p>
10565                            </a>
10566                            <a class="w3-text" href="lee24k_interspeech.html">
10567                                <p>
10568                                    Balanced-Wav2Vec: Enhancing Stability and Robustness of Representation Learning Through Sample Reweighting Techniques
10569                                    <br>
10570                                    <span class="w3-text w3-text-theme">
10571                                        Mun-Hak Lee, Jae-Hong Lee, DoHee Kim, Ye-Eun Ko, Joon-Hyuk Chang
10572                                    </span>
10573                                </p>
10574                            </a>
10575                        </div>
10576                    </div>
10577                    <br>
10578                    <div class="w3-content" style="height:10px"  id="Spoken Term Detection and Speech Retrieval"></div>
10579                    <div class="w3-card w3-round w3-white w3-padding">
10580                        <div class="w3-container"  style="margin-top:40px">
10581                            <h4 class="w3-center">Spoken Term Detection and Speech Retrieval</h4>
10582                            <hr>
10583                            <a class="w3-text" href="yuan24b_interspeech.html">
10584                                <p>
10585                                    Few-Shot Keyword Spotting from Mixed Speech
10586                                    <br>
10587                                    <span class="w3-text w3-text-theme">
10588                                        Junming Yuan, Ying Shi, LanTian Li, Dong Wang, Askar Hamdulla
10589                                    </span>
10590                                </p>
10591                            </a>
10592                            <a class="w3-text" href="yusuf24b_interspeech.html">
10593                                <p>
10594                                    Pretraining End-to-End Keyword Search with Automatically Discovered Acoustic Units
10595                                    <br>
10596                                    <span class="w3-text w3-text-theme">
10597                                        Bolaji Yusuf, Jan Honza Cernocky, Murat Saraçlar
10598                                    </span>
10599                                </p>
10600                            </a>
10601                            <a class="w3-text" href="he24_interspeech.html">
10602                                <p>
10603                                    2DP-2MRC: 2-Dimensional Pointer-based Machine Reading Comprehension Method for Multimodal Moment Retrieval
10604                                    <br>
10605                                    <span class="w3-text w3-text-theme">
10606                                        Jiajun He, Tomoki Toda
10607                                    </span>
10608                                </p>
10609                            </a>
10610                            <a class="w3-text" href="xie24c_interspeech.html">
10611                                <p>
10612                                    GPA: Global and Prototype Alignment for Audio-Text Retrieval
10613                                    <br>
10614                                    <span class="w3-text w3-text-theme">
10615                                        Yuxin Xie, Zhihong Zhu, Xianwei Zhuang, Liming Liang, Zhichang Wang, Yuexian Zou
10616                                    </span>
10617                                </p>
10618                            </a>
10619                            <a class="w3-text" href="kim24r_interspeech.html">
10620                                <p>
10621                                    Few-Shot Keyword-Incremental Learning with Total Calibration
10622                                    <br>
10623                                    <span class="w3-text w3-text-theme">
10624                                        Ilseok Kim, Ju-Seok Seong, Joon-Hyuk Chang
10625                                    </span>
10626                                </p>
10627                            </a>
10628                            <a class="w3-text" href="tapo24_interspeech.html">
10629                                <p>
10630                                    Leveraging Speech Data Diversity to Document Indigenous Heritage and Culture
10631                                    <br>
10632                                    <span class="w3-text w3-text-theme">
10633                                        Allahsera Tapo, Éric Le Ferrand, Zoey Liu, Christopher Homan, Emily Prud'hommeaux
10634                                    </span>
10635                                </p>
10636                            </a>
10637                        </div>
10638                    </div>
10639                    <br>
10640                    <div class="w3-content" style="height:10px"  id="Speech Disorders 1"></div>
10641                    <div class="w3-card w3-round w3-white w3-padding">
10642                        <div class="w3-container"  style="margin-top:40px">
10643                            <h4 class="w3-center">Speech Disorders 1</h4>
10644                            <hr>
10645                            <a class="w3-text" href="mohapatra24_interspeech.html">
10646                                <p>
10647                                    Missingness-resilient Video-enhanced Multimodal Disfluency Detection
10648                                    <br>
10649                                    <span class="w3-text w3-text-theme">
10650                                        Payal Mohapatra, Shamika Likhite, Subrata Biswas, Bashima Islam, Qi Zhu
10651                                    </span>
10652                                </p>
10653                            </a>
10654                            <a class="w3-text" href="gong24_interspeech.html">
10655                                <p>
10656                                    AS-70: A Mandarin stuttered speech dataset for automatic speech recognition and stuttering event detection
10657                                    <br>
10658                                    <span class="w3-text w3-text-theme">
10659                                        Rong Gong, Hongfei Xue, Lezhi Wang, Xin Xu, Qisheng Li, Lei Xie, Hui Bu, Shaomei Wu, Jiaming Zhou, Yong Qin, Binbin Zhang, Jun Du, Jia Bin, Ming Li
10660                                    </span>
10661                                </p>
10662                            </a>
10663                            <a class="w3-text" href="zulfikar24_interspeech.html">
10664                                <p>
10665                                    Analyzing Speech Motor Movement using Surface Electromyography in Minimally Verbal Adults with Autism Spectrum Disorder
10666                                    <br>
10667                                    <span class="w3-text w3-text-theme">
10668                                        Wazeer Zulfikar, Nishat Protyasha, Camila Canales, Heli Patel, James Williamson, Laura Sarnie, Lisa Nowinski, Nataliya Kosmyna, Paige Townsend, Sophia Yuditskaya, Tanya Talkar, Utkarsh Oggy Sarawgi, Christopher McDougle, Thomas Quatieri, Pattie Maes, Maria Mody
10669                                    </span>
10670                                </p>
10671                            </a>
10672                            <a class="w3-text" href="zhang24f_interspeech.html">
10673                                <p>
10674                                    Prosody of speech production in latent post-stroke aphasia
10675                                    <br>
10676                                    <span class="w3-text w3-text-theme">
10677                                        Cong Zhang, Tong Li, Gayle DeDe, Christos Salis
10678                                    </span>
10679                                </p>
10680                            </a>
10681                            <a class="w3-text" href="nie24_interspeech.html">
10682                                <p>
10683                                    MMSD-Net: Towards Multi-modal Stuttering Detection
10684                                    <br>
10685                                    <span class="w3-text w3-text-theme">
10686                                        Liangyu Nie, Sudarsana Reddy Kadiri, Ruchit Agrawal
10687                                    </span>
10688                                </p>
10689                            </a>
10690                            <a class="w3-text" href="wagner24b_interspeech.html">
10691                                <p>
10692                                    Large Language Models for Dysfluency Detection in Stuttered Speech
10693                                    <br>
10694                                    <span class="w3-text w3-text-theme">
10695                                        Dominik Wagner, Sebastian P. Bayerl, Ilja Baumann, Elmar Noeth, Korbinian Riedhammer, Tobias Bocklet
10696                                    </span>
10697                                </p>
10698                            </a>
10699                        </div>
10700                    </div>
10701                    <br>
10702                    <div class="w3-content" style="height:10px"  id="Connecting Speech-science and Speech-technology for Children’s Speech (Special Session)"></div>
10703                    <div class="w3-card w3-round w3-white w3-padding">
10704                        <div class="w3-container"  style="margin-top:40px">
10705                            <h4 class="w3-center">Connecting Speech-science and Speech-technology for Children’s Speech (Special Session)</h4>
10706                            <hr>
10707                            <a class="w3-text" href="demopoulos24_interspeech.html">
10708                                <p>
10709                                    Preliminary Investigation of Psychometric Properties of a Novel Multimodal Dialog Based Affect Production Task in Children and Adolescents with Autism
10710                                    <br>
10711                                    <span class="w3-text w3-text-theme">
10712                                        Carly Demopoulos, Linnea Lampinen, Cristian Preciado, Hardik Kothare, Vikram Ramanarayanan
10713                                    </span>
10714                                </p>
10715                            </a>
10716                            <a class="w3-text" href="charuau24_interspeech.html">
10717                                <p>
10718                                    Training speech-breathing coordination in computer-assisted reading
10719                                    <br>
10720                                    <span class="w3-text w3-text-theme">
10721                                        Delphine Charuau, Andrea Briglia, Erika Godde, Gérard Bailly
10722                                    </span>
10723                                </p>
10724                            </a>
10725                            <a class="w3-text" href="kadambi24_interspeech.html">
10726                                <p>
10727                                    How Does Alignment Error Affect Automated Pronunciation Scoring in Children's Speech?
10728                                    <br>
10729                                    <span class="w3-text w3-text-theme">
10730                                        Prad Kadambi, Tristan Mahr, Lucas Annear, Henry Nomeland, Julie Liss, Katherine Hustad, Visar Berisha
10731                                    </span>
10732                                </p>
10733                            </a>
10734                            <a class="w3-text" href="benway24_interspeech.html">
10735                                <p>
10736                                    Examining Vocal Tract Coordination in Childhood Apraxia of Speech with Acoustic-to-Articulatory Speech Inversion Feature Sets
10737                                    <br>
10738                                    <span class="w3-text w3-text-theme">
10739                                        Nina R. Benway, Jonathan L. Preston, Carol Espy-Wilson
10740                                    </span>
10741                                </p>
10742                            </a>
10743                            <a class="w3-text" href="sukhadia24_interspeech.html">
10744                                <p>
10745                                    Children’s Speech Recognition through Discrete Token Enhancement
10746                                    <br>
10747                                    <span class="w3-text w3-text-theme">
10748                                        Vrunda N. Sukhadia, Shammur Absar Chowdhury
10749                                    </span>
10750                                </p>
10751                            </a>
10752                            <a class="w3-text" href="wang24f_interspeech.html">
10753                                <p>
10754                                    Bridging Child-Centered Speech Language Identification and Language Diarization via Phonetics
10755                                    <br>
10756                                    <span class="w3-text w3-text-theme">
10757                                        Yujia Wang, Hexin Liu, Leibny Paola Garcia
10758                                    </span>
10759                                </p>
10760                            </a>
10761                            <a class="w3-text" href="gao24d_interspeech.html">
10762                                <p>
10763                                    Reading Miscue Detection in Primary School through Automatic Speech Recognition
10764                                    <br>
10765                                    <span class="w3-text w3-text-theme">
10766                                        Lingyun Gao, Cristian Tejedor-Garcia, Helmer Strik, Catia Cucchiarini
10767                                    </span>
10768                                </p>
10769                            </a>
10770                            <a class="w3-text" href="baumann24_interspeech.html">
10771                                <p>
10772                                    Automatic Evaluation of a Sentence Memory Test for Preschool Children
10773                                    <br>
10774                                    <span class="w3-text w3-text-theme">
10775                                        Ilja Baumann, Nicole Unger, Dominik Wagner, Korbinian Riedhammer, Tobias Bocklet
10776                                    </span>
10777                                </p>
10778                            </a>
10779                            <a class="w3-text" href="li24j_interspeech.html">
10780                                <p>
10781                                    Enhancing Child Vocalization Classification with  Phonetically-Tuned Embeddings for Assisting Autism Diagnosis
10782                                    <br>
10783                                    <span class="w3-text w3-text-theme">
10784                                        Jialu Li, Mark Hasegawa-Johnson, Karrie Karahalios
10785                                    </span>
10786                                </p>
10787                            </a>
10788                            <a class="w3-text" href="blockmedin24_interspeech.html">
10789                                <p>
10790                                    Self-Supervised Models for Phoneme Recognition: Applications in Children's Speech for Reading Learning
10791                                    <br>
10792                                    <span class="w3-text w3-text-theme">
10793                                        Lucas Block Medin, Thomas Pellegrini, Lucile Gelin
10794                                    </span>
10795                                </p>
10796                            </a>
10797                            <a class="w3-text" href="fan24b_interspeech.html">
10798                                <p>
10799                                    Benchmarking Children's ASR with Supervised and Self-supervised Speech Foundation Models
10800                                    <br>
10801                                    <span class="w3-text w3-text-theme">
10802                                        Ruchao Fan, Natarajan Balaji Shankar, Abeer Alwan
10803                                    </span>
10804                                </p>
10805                            </a>
10806                            <a class="w3-text" href="rolland24_interspeech.html">
10807                                <p>
10808                                    Introduction To Partial Fine-tuning: A Comprehensive Evaluation Of End-to-end Children’s Automatic Speech Recognition Adaptation
10809                                    <br>
10810                                    <span class="w3-text w3-text-theme">
10811                                        Thomas Rolland, Alberto Abad
10812                                    </span>
10813                                </p>
10814                            </a>
10815                            <a class="w3-text" href="zhang24d_interspeech.html">
10816                                <p>
10817                                    Improving child speech recognition with augmented child-like speech
10818                                    <br>
10819                                    <span class="w3-text w3-text-theme">
10820                                        Yuanyuan Zhang, Zhengjun Yue, Tanvina Patel, Odette Scharenborg
10821                                    </span>
10822                                </p>
10823                            </a>
10824                            <a class="w3-text" href="graave24_interspeech.html">
10825                                <p>
10826                                    Mixed Children/Adult/Childrenized Fine-Tuning for Children’s ASR: How to Reduce Age Mismatch and Speaking Style Mismatch
10827                                    <br>
10828                                    <span class="w3-text w3-text-theme">
10829                                        Thomas Graave, Zhengyang Li, Timo Lohrenz, Tim Fingscheidt
10830                                    </span>
10831                                </p>
10832                            </a>
10833                            <a class="w3-text" href="xu24c_interspeech.html">
10834                                <p>
10835                                    Exploring Speech Foundation Models for Speaker Diarization in Child-Adult Dyadic Interactions
10836                                    <br>
10837                                    <span class="w3-text w3-text-theme">
10838                                        Anfeng Xu, Kevin Huang, Tiantian Feng, Lue Shen, Helen Tager-Flusberg, Shrikanth Narayanan
10839                                    </span>
10840                                </p>
10841                            </a>
10842                        </div>
10843                    </div>
10844                    <br>
10845                    <div class="w3-content" style="height:10px"  id="Show and Tell 4"></div>
10846                    <div class="w3-card w3-round w3-white w3-padding">
10847                        <div class="w3-container"  style="margin-top:40px">
10848                            <h4 class="w3-center">Show and Tell 4</h4>
10849                            <hr>
10850                            <a class="w3-text" href="sirigiraju24_interspeech.html">
10851                                <p>
10852                                    IIITH Ucchar e-Sudharak: an automatic English pronunciation corrector for school-going children with a teacher in the loop
10853                                    <br>
10854                                    <span class="w3-text w3-text-theme">
10855                                        Meenakshi Sirigiraju, Arjun Rajasekar, Abhishikth Meejuri, Chiranjeevi Yarra
10856                                    </span>
10857                                </p>
10858                            </a>
10859                            <a class="w3-text" href="yap24_interspeech.html">
10860                                <p>
10861                                    Speech enabled visual acuity test
10862                                    <br>
10863                                    <span class="w3-text w3-text-theme">
10864                                        Boon Peng Yap, Kok Liang Tan, Zhenghao Li, Rong Tong
10865                                    </span>
10866                                </p>
10867                            </a>
10868                            <a class="w3-text" href="aiba24_interspeech.html">
10869                                <p>
10870                                    A ChatGPT-based oral Q&A practice system for first-time student participants in international conferences
10871                                    <br>
10872                                    <span class="w3-text w3-text-theme">
10873                                        Mayuko Aiba, Daisuke Saito, Nobuaki Minematsu
10874                                    </span>
10875                                </p>
10876                            </a>
10877                            <a class="w3-text" href="sridaran24_interspeech.html">
10878                                <p>
10879                                    Visual scene display application for augmentative and alternative communication
10880                                    <br>
10881                                    <span class="w3-text w3-text-theme">
10882                                        Karthik Venkat Sridaran, Raja Praveen, Reuben T Varghese, Ajish K Abraham, Shankar R, Winnie Rachel Cherian
10883                                    </span>
10884                                </p>
10885                            </a>
10886                            <a class="w3-text" href="masudakatsuse24_interspeech.html">
10887                                <p>
10888                                    CALL system using pitch-accent feature representations reflecting listeners’ subjective adequacy
10889                                    <br>
10890                                    <span class="w3-text w3-text-theme">
10891                                        Ikuyo Masuda-Katsuse, Ayako Shirose
10892                                    </span>
10893                                </p>
10894                            </a>
10895                            <a class="w3-text" href="preston24_interspeech.html">
10896                                <p>
10897                                    The speech motor chaining web app for speech motor learning
10898                                    <br>
10899                                    <span class="w3-text w3-text-theme">
10900                                        Jonathan L Preston, Nina R Benway, Nathan Prestopnik, Nathan Preston
10901                                    </span>
10902                                </p>
10903                            </a>
10904                            <a class="w3-text" href="yoder24_interspeech.html">
10905                                <p>
10906                                    Visualization for improving foreign language pronunciation
10907                                    <br>
10908                                    <span class="w3-text w3-text-theme">
10909                                        Charlotte Yoder, Karrie Karahalios, Mark Hasegawa-Johnson, Shreyansh Agrawal
10910                                    </span>
10911                                </p>
10912                            </a>
10913                            <a class="w3-text" href="phan24b_interspeech.html">
10914                                <p>
10915                                    CaptainA self-study mobile app for practising speaking: task completion assessment and feedback with generative AI
10916                                    <br>
10917                                    <span class="w3-text w3-text-theme">
10918                                        Nhan Phan, Anna von Zansen, Maria Kautonen, Tamás Grósz, Mikko Kurimo
10919                                    </span>
10920                                </p>
10921                            </a>
10922                        </div>
10923                    </div>
10924                    <br>
10925                </div>
10926            </div>
10927
10928
10929
10930            <!-- Paper search table -->
10931            <div class="w3-container" id="bypaper">
10932                <div class="w3-content" style="max-width:1200px;margin-top:60px">
10933                    <div class="w3-container w3-card w3-padding w3-white">
10934                        <div class="w3-text w3-center">
10935                            <span class='w3-large'>
10936                                <b>Search papers</b>
10937                            </span>
10938                            <button class='w3-text w3-button w3-right'
10939                                    onclick="document.getElementById('help_papers').style.display='block'">
10940                                <i class='icon-question-circle'></i>
10941                            </button>
10942                        </div>
10943                        <table id="paper_table" class="display" style="width:95%">
10944                            <thead>
10945                                <tr>
10946                                    <th width="100%">Article</th>
10947                                    <th width="0%"></th>
10948                                    <th width="0%"></th>
10949                                    <th width="0%"></th>
10950                                </tr>
10951                            </thead>
10952                        </table>
10953                    </div>
10954                    <!-- <p class="w3-small" style="margin-bottom: 50px"></p>   -->
10955                </div>
10956            </div>
10957        </div>
10958
10959        <!-- Session chooser -->
10960        <div id="sessionchooser" class="w3-modal" >
10961            <div class="w3-modal-content w3-card-4 w3-greyscale w3-theme-d4 w3-padding w3-bordered" onclick="document.getElementById('sessionchooser').style.display='none'">
10962                <span onclick="document.getElementById('sessionchooser').style.display='none'"
10963                      class="w3-button w3-display-topright">&times;</span>
10964                <p><a class="w3-text" href="#Keynote 1 ISCA Medallist">Keynote 1 ISCA Medallist</a></p>
10965                <p><a class="w3-text" href="#L2 Speech, Bilingualism and Code-Switching">L2 Speech, Bilingualism and Code-Switching</a></p>
10966                <p><a class="w3-text" href="#Speaker Diarization 1">Speaker Diarization 1</a></p>
10967                <p><a class="w3-text" href="#Speech and Audio Analysis and Representations">Speech and Audio Analysis and Representations</a></p>
10968                <p><a class="w3-text" href="#Acoustic Event Detection and Classification 2">Acoustic Event Detection and Classification 2</a></p>
10969                <p><a class="w3-text" href="#Detection and Classification of Bioacoustic Signals">Detection and Classification of Bioacoustic Signals</a></p>
10970                <p><a class="w3-text" href="#Acoustic Echo Cancellation">Acoustic Echo Cancellation</a></p>
10971                <p><a class="w3-text" href="#Speech Synthesis: Voice Conversion 1">Speech Synthesis: Voice Conversion 1</a></p>
10972                <p><a class="w3-text" href="#Neural Network Architectures for ASR 2">Neural Network Architectures for ASR 2</a></p>
10973                <p><a class="w3-text" href="#Decoding Algorithms">Decoding Algorithms</a></p>
10974                <p><a class="w3-text" href="#Pronunciation Assessment">Pronunciation Assessment</a></p>
10975                <p><a class="w3-text" href="#Spoken Language Processing">Spoken Language Processing</a></p>
10976                <p><a class="w3-text" href="#Spoken Machine Translation 2">Spoken Machine Translation 2</a></p>
10977                <p><a class="w3-text" href="#Biosignal-enabled Spoken Communication">Biosignal-enabled Spoken Communication</a></p>
10978                <p><a class="w3-text" href="#Individual and Social Factors in Phonetics">Individual and Social Factors in Phonetics</a></p>
10979                <p><a class="w3-text" href="#Paralinguistics">Paralinguistics</a></p>
10980                <p><a class="w3-text" href="#Speaker Recognition: Adversarial and Spoofing Attacks">Speaker Recognition: Adversarial and Spoofing Attacks</a></p>
10981                <p><a class="w3-text" href="#Audio Event Detection and Classification 1">Audio Event Detection and Classification 1</a></p>
10982                <p><a class="w3-text" href="#Source Separation 2">Source Separation 2</a></p>
10983                <p><a class="w3-text" href="#Noise Reduction, Dereverberation, and Echo Cancellation">Noise Reduction, Dereverberation, and Echo Cancellation</a></p>
10984                <p><a class="w3-text" href="#Computationally-Efficient Speech Enhancement">Computationally-Efficient Speech Enhancement</a></p>
10985                <p><a class="w3-text" href="#Zero-shot TTS">Zero-shot TTS</a></p>
10986                <p><a class="w3-text" href="#Noise Robustness, Far-Field, and Multi-Talker ASR">Noise Robustness, Far-Field, and Multi-Talker ASR</a></p>
10987                <p><a class="w3-text" href="#Contextual Biasing and Adaptation">Contextual Biasing and Adaptation</a></p>
10988                <p><a class="w3-text" href="#Spoken Language Understanding">Spoken Language Understanding</a></p>
10989                <p><a class="w3-text" href="#Spoken Machine Translation 1">Spoken Machine Translation 1</a></p>
10990                <p><a class="w3-text" href="#Hearing Disorders">Hearing Disorders</a></p>
10991                <p><a class="w3-text" href="#Speech Disorders 2">Speech Disorders 2</a></p>
10992                <p><a class="w3-text" href="#TAUKADIAL Challenge: Speech-Based Cognitive Assessment in Chinese and English (Special Session)">TAUKADIAL Challenge: Speech-Based Cognitive Assessment in Chinese and English (Special Session)</a></p>
10993                <p><a class="w3-text" href="#Show and Tell 1">Show and Tell 1</a></p>
10994                <p><a class="w3-text" href="#Keynote 2">Keynote 2</a></p>
10995                <p><a class="w3-text" href="#Phonetics and Phonology of Second Language Acquisition">Phonetics and Phonology of Second Language Acquisition</a></p>
10996                <p><a class="w3-text" href="#Corpora-based Approaches in Automatic Emotion Recognition">Corpora-based Approaches in Automatic Emotion Recognition</a></p>
10997                <p><a class="w3-text" href="#Analysis of Speakers States and Traits">Analysis of Speakers States and Traits</a></p>
10998                <p><a class="w3-text" href="#Spoofing and Deepfake Detection">Spoofing and Deepfake Detection</a></p>
10999                <p><a class="w3-text" href="#Audio Captioning, Tagging, and Audio-Text Retrieval">Audio Captioning, Tagging, and Audio-Text Retrieval</a></p>
11000                <p><a class="w3-text" href="#Generative Speech Enhancement">Generative Speech Enhancement</a></p>
11001                <p><a class="w3-text" href="#Speech Synthesis: Evaluation">Speech Synthesis: Evaluation</a></p>
11002                <p><a class="w3-text" href="#Multilingual ASR">Multilingual ASR</a></p>
11003                <p><a class="w3-text" href="#General Topics in ASR">General Topics in ASR</a></p>
11004                <p><a class="w3-text" href="#Spoken Language Understanding">Spoken Language Understanding</a></p>
11005                <p><a class="w3-text" href="#Speech and Multimodal Resources">Speech and Multimodal Resources</a></p>
11006                <p><a class="w3-text" href="#Pathological Speech Analysis 1">Pathological Speech Analysis 1</a></p>
11007                <p><a class="w3-text" href="#Speech and Language in Health: from Remote Monitoring to Medical Conversations - 1 (Special Session)">Speech and Language in Health: from Remote Monitoring to Medical Conversations - 1 (Special Session)</a></p>
11008                <p><a class="w3-text" href="#Speech and Brain">Speech and Brain</a></p>
11009                <p><a class="w3-text" href="#Innovative Methods in Phonetics and Phonology">Innovative Methods in Phonetics and Phonology</a></p>
11010                <p><a class="w3-text" href="#Voice, Tones and F0">Voice, Tones and F0</a></p>
11011                <p><a class="w3-text" href="#Emotion Recognition: Resources and Benchmarks">Emotion Recognition: Resources and Benchmarks</a></p>
11012                <p><a class="w3-text" href="#Speaker and Language Identification and Diarization">Speaker and Language Identification and Diarization</a></p>
11013                <p><a class="w3-text" href="#Audio-Text Retrieval">Audio-Text Retrieval</a></p>
11014                <p><a class="w3-text" href="#Speech Enhancement">Speech Enhancement</a></p>
11015                <p><a class="w3-text" href="#Speech Coding">Speech Coding</a></p>
11016                <p><a class="w3-text" href="#Speech Synthesis: Expressivity and Emotion">Speech Synthesis: Expressivity and Emotion</a></p>
11017                <p><a class="w3-text" href="#Speech Synthesis: Tools and Data">Speech Synthesis: Tools and Data</a></p>
11018                <p><a class="w3-text" href="#Speech Synthesis: Singing Voice Synthesis">Speech Synthesis: Singing Voice Synthesis</a></p>
11019                <p><a class="w3-text" href="#LLM in ASR">LLM in ASR</a></p>
11020                <p><a class="w3-text" href="#Vision and Speech">Vision and Speech</a></p>
11021                <p><a class="w3-text" href="#Spoken Document Summarization">Spoken Document Summarization</a></p>
11022                <p><a class="w3-text" href="#Speech and Language in Health: from Remote Monitoring to Medical Conversations - 2 (Special Sessions)">Speech and Language in Health: from Remote Monitoring to Medical Conversations - 2 (Special Sessions)</a></p>
11023                <p><a class="w3-text" href="#Show and Tell 2">Show and Tell 2</a></p>
11024                <p><a class="w3-text" href="#Prosody">Prosody</a></p>
11025                <p><a class="w3-text" href="#Foundational Models for Deepfake and Spoofed Speech Detection">Foundational Models for Deepfake and Spoofed Speech Detection</a></p>
11026                <p><a class="w3-text" href="#Speaker Recognition 1">Speaker Recognition 1</a></p>
11027                <p>
11027<a class="w3-text" href="#Source Separation 1">Source Separation 1</a></p>
11028                <p><a class="w3-text" href="#Audio-Visual and Generative Speech Enhancement">Audio-Visual and Generative Speech Enhancement</a></p>
11029                <p><a class="w3-text" href="#Speech Privacy and Bandwidth Expansion">Speech Privacy and Bandwidth Expansion</a></p>
11030                <p><a class="w3-text" href="#Speech Synthesis: Prosody">Speech Synthesis: Prosody</a></p>
11031                <p><a class="w3-text" href="#Accented Speech, Prosodic Features, Dialect, Emotion, Sound Classification">Accented Speech, Prosodic Features, Dialect, Emotion, Sound Classification</a></p>
11032                <p><a class="w3-text" href="#Neural Network Adaptation">Neural Network Adaptation</a></p>
11033                <p><a class="w3-text" href="#ASR and LLMs">ASR and LLMs</a></p>
11034                <p><a class="w3-text" href="#Pathological Speech Analysis 3">Pathological Speech Analysis 3</a></p>
11035                <p><a class="w3-text" href="#Speech Disorders 3">Speech Disorders 3</a></p>
11036                <p><a class="w3-text" href="#Speech Recognition with Large Pretrained Speech Models for Under-represented Languages (Special Session)">Speech Recognition with Large Pretrained Speech Models for Under-represented Languages (Special Session)</a></p>
11037                <p><a class="w3-text" href="#Speech Processing Using Discrete Speech Units (Special Session)">Speech Processing Using Discrete Speech Units (Special Session)</a></p>
11038                <p><a class="w3-text" href="#Keynote 3">Keynote 3</a></p>
11039                <p><a class="w3-text" href="#Databases and Progress in Methodology">Databases and Progress in Methodology</a></p>
11040                <p><a class="w3-text" href="#Articulation, Convergence and Perception">Articulation, Convergence and Perception</a></p>
11041                <p><a class="w3-text" href="#Speech Emotion Recognition">Speech Emotion Recognition</a></p>
11042                <p><a class="w3-text" href="#Self-Supervised Models in Speaker Recognition">Self-Supervised Models in Speaker Recognition</a></p>
11043                <p><a class="w3-text" href="#Speech Quality Assessment">Speech Quality Assessment</a></p>
11044                <p><a class="w3-text" href="#Privacy and Security in Speech Communication 1">Privacy and Security in Speech Communication 1</a></p>
11045                <p><a class="w3-text" href="#Speech Synthesis: Voice Conversion 2">Speech Synthesis: Voice Conversion 2</a></p>
11046                <p><a class="w3-text" href="#Speech Synthesis: Text Processing">Speech Synthesis: Text Processing</a></p>
11047                <p><a class="w3-text" href="#Training Methods, Self-Supervised Learning, Adaptation">Training Methods, Self-Supervised Learning, Adaptation</a></p>
11048                <p><a class="w3-text" href="#Novel Architectures for ASR">Novel Architectures for ASR</a></p>
11049                <p><a class="w3-text" href="#Multimodality and Foundation Models">Multimodality and Foundation Models</a></p>
11050                <p><a class="w3-text" href="#Spoken Dialogue Systems and Conversational Analysis 1">Spoken Dialogue Systems and Conversational Analysis 1</a></p>
11051                <p><a class="w3-text" href="#Speech Technology">Speech Technology</a></p>
11052                <p><a class="w3-text" href="#Pathological Speech Analysis 2">Pathological Speech Analysis 2</a></p>
11053                <p><a class="w3-text" href="#Speech Science, Speech Technology, and Gender (Special Session)">Speech Science, Speech Technology, and Gender (Special Session)</a></p>
11054                <p><a class="w3-text" href="#Speech Production and Perception">Speech Production and Perception</a></p>
11055                <p><a class="w3-text" href="#Phonetics and Phonology: Segmentals and Suprasegmentals">Phonetics and Phonology: Segmentals and Suprasegmentals</a></p>
11056                <p><a class="w3-text" href="#Topics in Paralinguistics">Topics in Paralinguistics</a></p>
11057                <p><a class="w3-text" href="#Emotion Recognition: Fairness, Variability, Uncertainty">Emotion Recognition: Fairness, Variability, Uncertainty</a></p>
11058                <p><a class="w3-text" href="#Speaker Verification">Speaker Verification</a></p>
11059                <p><a class="w3-text" href="#Spatial Audio and Acoustics">Spatial Audio and Acoustics</a></p>
11060                <p><a class="w3-text" href="#Generative Models for Speech and Audio">Generative Models for Speech and Audio</a></p>
11061                <p><a class="w3-text" href="#Speech and Audio Modelling">Speech and Audio Modelling</a></p>
11062                <p><a class="w3-text" href="#Multi-Channel Speech Enhancement">Multi-Channel Speech Enhancement</a></p>
11063                <p><a class="w3-text" href="#Speech Synthesis: Paradigms and Methods 1">Speech Synthesis: Paradigms and Methods 1</a></p>
11064                <p><a class="w3-text" href="#Speech Synthesis: Paradigms and Methods 2">Speech Synthesis: Paradigms and Methods 2</a></p>
11065                <p><a class="w3-text" href="#Neural Network Architectures for ASR 1">Neural Network Architectures for ASR 1</a></p>
11066                <p><a class="w3-text" href="#Error Correction and Rescoring">Error Correction and Rescoring</a></p>
11067                <p><a class="w3-text" href="#Spoken Language Understanding">Spoken Language Understanding</a></p>
11068                <p><a class="w3-text" href="#Spoken Dialogue Systems and Conversational Analysis 2">Spoken Dialogue Systems and Conversational Analysis 2</a></p>
11069                <p><a class="w3-text" href="#Computational Models of Human Language Acquisition, Perception, and Production (Special Session)">Computational Models of Human Language Acquisition, Perception, and Production (Special Session)</a></p>
11070                <p><a class="w3-text" href="#Show and Tell 3">Show and Tell 3</a></p>
11071                <p><a class="w3-text" href="#Phonetics, Phonology and Prosody">Phonetics, Phonology and Prosody</a></p>
11072                <p><a class="w3-text" href="#Segmentals">Segmentals</a></p>
11073                <p><a class="w3-text" href="#New Avenues in Emotion Recognition">New Avenues in Emotion Recognition</a></p>
11074                <p><a class="w3-text" href="#Speaker Diarization 2">Speaker Diarization 2</a></p>
11075                <p><a class="w3-text" href="#Speaker Recognition 2">Speaker Recognition 2</a></p>
11076                <p><a class="w3-text" href="#Speech and Audio Analysis">Speech and Audio Analysis</a></p>
11077                <p><a class="w3-text" href="#Speech Quality and Intelligibility: Prediction and Enhancement">Speech Quality and Intelligibility: Prediction and Enhancement</a></p>
11078                <p><a class="w3-text" href="#Speech Synthesis: Vocoders">Speech Synthesis: Vocoders</a></p>
11079                <p><a class="w3-text" href="#ASR Model Training Methods">ASR Model Training Methods</a></p>
11080                <p><a class="w3-text" href="#Cross-Lingual and Multilingual Processing">Cross-Lingual and Multilingual Processing</a></p>
11081                <p><a class="w3-text" href="#Speech Assessment">Speech Assessment</a></p>
11082                <p><a class="w3-text" href="#Question Answering from Speech and Spoken Dialogue Systems">Question Answering from Speech and Spoken Dialogue Systems</a></p>
11083                <p><a class="w3-text" href="#Spoken Dialogue Systems and Conversational Analysis 3">Spoken Dialogue Systems and Conversational Analysis 3</a></p>
11084                <p><a class="w3-text" href="#Dysarthric Speech Assessment">Dysarthric Speech Assessment</a></p>
11085                <p><a class="w3-text" href="#Spoken Language Models for Universal Speech Processing (Special Session)">Spoken Language Models for Universal Speech Processing (Special Session)</a></p>
11086                <p><a class="w3-text" href="#Keynote 4">Keynote 4</a></p>
11087                <p><a class="w3-text" href="#L1/L2 Acquisition and Cross-Linguistic Factors">L1/L2 Acquisition and Cross-Linguistic Factors</a></p>
11088                <p><a class="w3-text" href="#Speaker Stance, Emotion and Language-External Factors">Speaker Stance, Emotion and Language-External Factors</a></p>
11089                <p><a class="w3-text" href="#Experimental Phonetics and Laboratory Phonology">Experimental Phonetics and Laboratory Phonology</a></p>
11090                <p><a class="w3-text" href="#Speaker recognition evaluation and resources">Speaker recognition evaluation and resources</a></p>
11091                <p><a class="w3-text" href="#Speech Type Classification">Speech Type Classification</a></p>
11092                <p><a class="w3-text" href="#Target Speaker Extraction">Target Speaker Extraction</a></p>
11093                <p><a class="w3-text" href="#Speech Synthesis: Voice Conversion 3">Speech Synthesis: Voice Conversion 3</a></p>
11094                <p><a class="w3-text" href="#Speech Synthesis: Paradigms and Methods 3">Speech Synthesis: Paradigms and Methods 3</a></p>
11095                <p><a class="w3-text" href="#Privacy and Security in Speech Communication 2">Privacy and Security in Speech Communication 2</a></p>
11096                <p><a class="w3-text" href="#Streaming ASR">Streaming ASR</a></p>
11097                <p><a class="w3-text" href="#Computational Resource Constrained ASR">Computational Resource Constrained ASR</a></p>
11098                <p><a class="w3-text" href="#Evaluation of Speech Technology Systems">Evaluation of Speech Technology Systems</a></p>
11099                <p><a class="w3-text" href="#Neural Network Training for Speech Recognition">Neural Network Training for Speech Recognition</a></p>
11100                <p><a class="w3-text" href="#Leveraging Large Language Models and Contextual Features for Phonetic Analysis (Special Session)">Leveraging Large Language Models and Contextual Features for Phonetic Analysis (Special Session)</a></p>
11101                <p><a class="w3-text" href="#Responsible Speech Foundation Models (Special Session)">Responsible Speech Foundation Models (Special Session)</a></p>
11102                <p><a class="w3-text" href="#Multimodal Paralinguistics">Multimodal Paralinguistics</a></p>
11103                <p><a class="w3-text" href="#Automatic Emotion Recognition">Automatic Emotion Recognition</a></p>
11104                <p><a class="w3-text" href="#Self and Weakly-Labelled Speaker Verification">Self and Weakly-Labelled Speaker Verification</a></p>
11105                <p><a class="w3-text" href="#Acoustic Event Detection, Segmentation and Classification">Acoustic Event Detection, Segmentation and Classification</a></p>
11106                <p><a class="w3-text" href="#Speech and Audio Modelling">Speech and Audio Modelling</a></p>
11107                <p><a class="w3-text" href="#Fake Audio Detection">Fake Audio Detection</a></p>
11108                <p><a class="w3-text" href="#Deep Learning-Based Speech Enhancement: Approaches, Scalability, and Evaluation">Deep Learning-Based Speech Enhancement: Approaches, Scalability, and Evaluation</a></p>
11109                <p><a class="w3-text" href="#Speech Synthesis: Other Topics 1">Speech Synthesis: Other Topics 1</a></p>
11110                <p><a class="w3-text" href="#Speech Synthesis: Other Topics 2">Speech Synthesis: Other Topics 2</a></p>
11111                <p><a class="w3-text" href="#Speech synthesis: Cross-lingual and multilingual aspects">Speech synthesis: Cross-lingual and multilingual aspects</a></p>
11112                <p><a class="w3-text" href="#Noise, Far-Field, Multi-Talker, Enhancement, Audio Classification">Noise, Far-Field, Multi-Talker, Enhancement, Audio Classification</a></p>
11113                <p><a class="w3-text" href="#Self-Supervised Learning for ASR">Self-Supervised Learning for ASR</a></p>
11114                <p><a class="w3-text" href="#Spoken Term Detection and Speech Retrieval">Spoken Term Detection and Speech Retrieval</a></p>
11115                <p><a class="w3-text" href="#Speech Disorders 1">Speech Disorders 1</a></p>
11116                <p><a class="w3-text" href="#Connecting Speech-science and Speech-technology for Children’s Speech (Special Session)">Connecting Speech-science and Speech-technology for Children’s Speech (Special Session)</a></p>
11117                <p><a class="w3-text" href="#Show and Tell 4">Show and Tell 4</a></p>
11118            </div>
11119        </div>
11120
11121
11122        
11122<script>
11123            function myFunction() {
11124                var x = document.getElementById("smallnav");
11125                if (x.className.indexOf("w3-show") == -1) {
11126                    x.className += " w3-show";
11127                } else {
11128                    x.className = x.className.replace(" w3-show", "");
11129                }
11130            }
11131
11132            // Get the modal
11133            var modal = document.getElementById('sessionchooser');
11134
11135            // When the user clicks anywhere outside of the modal, close it
11136            window.onclick = function(event) {
11137                if (event.target == modal) {
11138                    modal.style.display = "none";
11139                }
11140            }
11141
11142
11143            $(document).ready(function() {
11144
11145                $('#paper_table').DataTable( {
11146                    data: [['Zhiqi Ai, Zhiyong Chen, Shugong Xu', 'MM-KWS: Multi-modal Prompts for Multilingual User-defined Keyword Spotting', 'ai24_interspeech', 'text embeddings libriphrase speech mining confusable enrollment applicability distinguishing solely'], ['Ye-Xin Lu, Yang Ai, Zheng-Yan Sheng, Zhen-Hua Ling', 'MultiStage Speech Bandwidth Extension with Flexible Sampling Rate Control', 'lu24_interspeech', 'ms-bwe bwe block generation stage sixty one-stage featuring source multi-stage'], ['Olympia Simantiraki, Martin Cooke', "Listeners' F0 preferences in quiet and stationary noise", 'simantiraki24_interspeech', 'alter mean intelligibility impact original inexperienced maximise comprehensibility variation masker'], ['Jennifer Williams, Eike Schneiders, Henry Card, Tina Seabrooke, Beatrice Pakenham-Walsh, Tayyaba Azim, Lucy Valls-Reed, Ganesh Vigneswaran, John Robert Bautista, Rohan Chandra, Arya Farahi', 'Predicting Acute Pain Levels Implicitly from Vocal Features', 'williams24_interspeech', 'clinical triage high-stakes three-class communication explainable cold stroke support emergency'], ['Xingxing Yang', 'G2PA: G2P with Aligned Audio for Mandarin Chinese', 'yang24_interspeech', 'pronunciation github solely ambiguity phoneme text disregarding preprocess polyphone repository'], ['Sizhou Chen, Yibo Bai, Jiadi Yao, Xiao-Lei Zhang, Xuelong Li', 'Textual-Driven Adversarial Purification for Speaker Verification', 'chen24_interspeech', 'diffusion audio defense attack textual neglecting diffusion-based model causing mistake'], ['Thien-Phuc Doan, Long Nguyen-Vu, Kihun Hong, Souhwan Jung', 'Balance, Multiple Augmentation, and Re-synthesis: A Triad Training Strategy for Enhanced Audio Deepfake Detection', 'doan24_interspeech', 'assembling surpassed set sample mini-batch in-the-wild re-synthesized benchmarking balancing achievable'], ['Mohan Li, Simon Keizer, Rama Doddipatla', 'Prompting Whisper for QA-driven Zero-shot End-to-end Spoken Language Understanding', 'li24_interspeech', 'slu slurp question-answering system optimising comprehend model cross-corpus excessive comparably'], ['Daisuke Niizumi, Daiki Takeuchi, Yasunori Ohishi, Noboru Harada, Masahiro Yasuda, Shunsuke Tsubaki, Keisuke Imoto', 'M2D-CLAP: Masked Modeling Duo Meets CLAP for Learning General-purpose Audio-Language Representation', 'niizumi24_interspeech', 'learns audio transfer performs gtzan versatile aligns classification zero-shot pre-training'], ['Nicolas M. Müller, Piotr Kawa, Shen Hu, Matthias Neu, Jennifer Williams, Philip Sperl, Konstantin Böttinger', 'A New Approach to Voice Authenticity', 'muller24_interspeech', 'edits fake editing binary maliciously ethically pinpointing nancy longstanding delineate'], ['Korbinian Kuhn, Verena Kersken, Gottfried Zimmermann', 'Beyond Levenshtein: Leveraging Multiple Algorithms for Robust Word Error Rate Computations And Granular Error Classifications', 'kuhn24_interspeech', 'punctuation wer non-semantic token-based exemplary common pre-processed visualisation substituting equivalence'], ['Jiatong Shi, Yueqian Lin, Xinyi Bai, Keyi Zhang, Yuning Wu, Yuxun Tang, Yifeng Yu, Qin Jin, Shinji Watanabe', 'Singing Voice Data Scaling-up: An Introduction to ACE-Opencpop and ACE-KiSing', 'shi24_interspeech', 'svs espnet datasets synthesis expansive multi-singer curation complemented supplementary thorough'], ['Umberto Cappellazzo, Daniele Falavigna, Alessio Brutti', 'Efficient Fine-tuning of Audio Spectrogram Transformers via Soft Mixture of Adapters', 'cappellazzo24_interspeech', 'moe expert burgeoning computational parameter-efficient underexplored affordable adapter dense ablation'], ['Titouan Parcollet, Rogier van Dalen, Shucong Zhang, Sourav Bhattacharya', 'SummaryMixing: A Linear-Complexity Alternative to Self-Attention for Speech Recognition and Understanding', 'parcollet24_interspeech', 'quot inference memory cheaper summarises slowing asr exceed quadratic consumption'], ['Soham Deshmukh, Rita Singh, Bhiksha Raj', 'Domain Adaptation for Contrastive Audio-Language Models', 'deshmukh24_interspeech', 'alm zero-shot alms prompt generalization access require performance test-time enforcing'], ['Jan Pešán, Vojtěch Juřík, Martin Karafiát, Jan Černocký', 'BESST Dataset: A Multimodal Resource for Speech-based Stress Detection and Analysis', 'pesan24_interspeech', 'comprises speech physiological collection electrocardiogram data electrodermal clean temperature immersion'], ['Martijn Bentum, Louis ten Bosch, Tom Lentz', 'The Processing of Stress in End-to-End Automatic Speech Recognition Models', 'bentum24_interspeech', 'classifier correlate denominator layer acoustic mere vowel asr reflection representation'], ['Zhouyuan Huo, Dongseong Hwang, Gan Song, Khe Chai Sim, Weiran Wang', 'AdaRA: Adaptive Rank Allocation of Residual Adapters for Speech Foundation Model', 'huo24_interspeech', 'parameter additive adaptation bottleneck dimension o
11146ptimal allocating layer possessing efficient'], ['Xinlei Niu, Jing Zhang, Charles Patrick Martin', 'HybridVC: Efficient Voice Style Conversion with Text and Audio Prompts', 'niu24_interspeech', 'contrastive latent embeddings cvae optimises user-defined personalised validates speaker underscore'], ['Oleg Rybakov, Dmitriy Serdyuk, Chengjian Zheng', 'USM RNN-T model weights binarization', 'rybakov24_interspeech', 'size grows serving cost float reduction dominated attractive transducer quantization'], ['Ji-Sang Hwang, Hyeongrae Noh, Yoonseok Hong, Insoo Oh', 'X-Singer: Code-Mixed Singing Voice Synthesis via Cross-Lingual Learning', 'hwang24_interspeech', 'svs lyric annotation phoneme synthesize encoder mixture ability code-switching intra'], ['Yingying Gao, Shilei Zhang, Chao Deng, Junlan Feng', 'GenDistiller: Distilling Pre-trained Language Models based on an Autoregressive Generative Model', 'gao24_interspeech', 'superb wavlm teacher autoregressively resource layer-by-layer hidden hinder hubert distillation'], ['Rui Cao, Tianrui Wang, Meng Ge, Andong Li, Longbiao Wang, Jianwu Dang, Yungang Jia', 'VoiCor: A Residual Iterative Voice Correction Framework for Monaural Speech Enhancement', 'cao24_interspeech', 'dns-challenge solution non-linearly concretely issue structure continue chain refine pesq'], ['Heeseung Kim, Sang-gil Lee, Jiheum Yeom, Che Hyun Lee, Sungwon Kim, Sungroh Yoon', 'VoiceTailor: Lightweight Plug-In Adapter for Diffusion-Based Personalized Text-to-Speech', 'kim24_interspeech', 'adaptation parameter-efficient pivotal tt speaker pre-trained equipping module lora speaker-adaptive'], ['Jizhong Liu, Gang Li, Junbo Zhang, Heinrich Dinkel, Yongqing Wang, Zhiyong Yan, Yujun Wang, Bin Wang', 'Enhancing Automated Audio Captioning via Large Language Models with Optimized Audio Encoding', 'liu24_interspeech', 'aac llm token pre-trained encoder decoder effectivity ced llama querying'], ['Siqi Sun, Korin Richmond', 'Learning Pronunciation from Other Accents via Pronunciation Knowledge Transfer', 'sun24_interspeech', 'seq word target bootstrapping frontend annotating transferred accent coverage type'], ['Masaya Ohagi, Tomoya Mizumoto, Katsumasa Yoshikawa', 'Investigation of look-ahead techniques to improve response time in spoken dialogue system', 'ohagi24_interspeech', 'user speed utterance chatbot bot finish returned delayed task-oriented say'], ['Si Chen, Bruce Xiao Wang, Yitian Hong, Fang Zhou, Angel Chan, Po-yi Tang, Bin Li, Chunyi Wen, James Cheung, Yan Liu, Zhuoming Chen', 'Acoustic changes in speech prosody produced by children with autism after robot-assisted speech training', 'chen24b_interspeech', 'autistic marking focus typically-developing variability monotone post-training designed mixed-effects signalling'], ['Wenbin Wang, Yang Song, Sanjay Jha', 'GLOBE: A High-quality English Corpus with Global Accents for Zero-shot Speaker Adaptive Text-to-Speech', 'wang24b_interspeech', 'worldwide metadata populating tt libritts curated rigorous vctk generalizability trained'], ['Jaden Pieper, Stephen Voran', 'AlignNet: Learning dataset score alignment functions to enable better training of speech quality estimators', 'pieper24_interspeech', 'mdf datasets aligner estimator us intermediate successful larger pretrains no-reference'], ['Ji Won Yoon, Beom Jun Woo, Nam Soo Kim', 'HuBERT-EE: Early Exiting HuBERT for Efficient Speech Recognition', 'yoon24_interspeech', 'exit inference branch intermediate stop model hidden-unit returned slowing confident'], ['Hideyuki Oiso, Yuto Matsunaga, Kazuya Kakizaki, Taiki Miyagawa', 'Prompt Tuning for Audio Deepfake Detection: Computationally Efficient Test-time Domain Adaptation with Limited Target Dataset', 'oiso24_interspeech', 'add computational gap challenge extra iii countering cost source-target plug-in'], ['Bunlong Lay, Timo Gerkmann', 'An Analysis of the Variance of Diffusion-based Speech Enhancement', 'lay24_interspeech', 'diffusion attenuation noise equation differential environmental stochastic adding gaussian concretely'], ['Lukas Christ, Shahin Amiriparian, Friederike Hawighorst, Ann-Kathrin Schill, Angelo Boutalikakis, Lorenz Graf-Vlachy, Andreas König, Björn Schuller', 'This Paper Had the Smartest Reviewers 
11146- Flattery Detection Utilising an Audio-Textual Transformer-Based Approach', 'christ24_interspeech', 'whisper textual multimodal modality praise compliment bonding build roberta ast'], ['Zengrui Jin, Yifan Yang, Mohan Shi, Wei Kang, Xiaoyu Yang, Zengwei Yao, Fangjun Kuang, Liyong Guo, Lingwei Meng, Long Lin, Yong Xu, Shi-Xiong Zhang, Daniel Povey', 'LibriheavyMix: A 20,000-Hour Dataset for Single-Channel Reverberant Multi-Talker Speech Separation, ASR and Speaker Diarization', 'jin24_interspeech', 'far-field daunting multitalker landscape foundational crafted convenience generality encompassing challenge'], ['Andreas Triantafyllopoulos, Anton Batliner, Wolfgang Mayr, Markus Fendler, Florian Pokorny, Maurice Gerczuk, Shahin Amiriparian, Thomas Berghaus, Björn Schuller', 'Sustained Vowels for Pre- vs Post-Treatment COPD Classification', 'triantafyllopoulos24_interspeech', 'lung disease patient exacerbation obstructed hospitalisation pulmonary read acute obstructive'], ['Andreas Triantafyllopoulos, Anton Batliner, Simon Rampp, Manuel Milling, Björn Schuller', 'INTERSPEECH 2009 Emotion Challenge Revisited: Benchmarking 15 Years of Progress in Speech Emotion Recognition', 'triantafyllopoulos24b_interspeech', 'ser official outperform corollary -the set winner newer hyperparameter model'], ['Andreas Triantafyllopoulos, Björn Schuller', 'Enrolment-based personalisation for improving individual-level fairness in speech emotion recognition', 'triantafyllopoulos24c_interspeech', 'aggregated evaluation ser one-size-fits-all individualistic failing uncovered enrolment speaker across'], ['Zhenyu Wang, Shuyu Kong, Li Wan, Biqiao Zhang, Yiteng Huang, Mumin Jin, Ming Sun, Xin Lei, Zhaojun Yang', 'Query-by-Example Keyword Spotting Using Spectral-Temporal Graph Attentive Pooling and Multi-Task Learning', 'wang24c_interspeech', 'liconet qbye kw conformer framework fa tailoring speaker-invariant frr customized'], ['Daniela A. Wiepert, Rene L. Utianski, Joseph R. Duffy, John L. Stricker, Leland R. Barnard, David T. Jones, Hugo Botha', 'Speech foundation models in healthcare: Effect of layer selection on pathological speech feature prediction', 'wiepert24_interspeech', 'diagnosis treatment clinical best out-of-distribution lower average increase worst neurological'], ['Yuang Li, Jiawei Yu, Min Zhang, Mengxin Ren, Yanqing Zhao, Xiaofeng Zhao, Shimin Tao, Jinsong Su, Hao Yang', 'Using Large Language Model for End-to-End Chinese ASR and NER', 'li24b_interspeech', 'decoder-only llm encoder-decoder token long-form speech infers connect taxonomy cross-attention'], ['Yuang Li, Min Zhang, Chang Su, Yinglu Li, Xiaosong Qiao, Mengxin Ren, Miaomiao Ma, Daimeng Wei, Shimin Tao, Hao Yang', 'A Multitask Training Approach to Enhance Whisper with Open-Vocabulary Keyword Spotting', 'li24c_interspeech', 'ov-kws entity asr named aishell plug-and-play user-defined terminology model hot'], ['Li Xiao, Lucheng Fang, Yuhong Yang, Weiping Tu', 'LungAdapter: Efficient Adapting Audio Spectrogram Transformer for Lung Sound Classification', 'xiao24_interspeech', 'fine-tuning large-scale pre-trained model full parameter adapter predominant frozen entail'], ['Yi Gao, Xiang Su', 'Low Complexity Echo Delay Estimator Based on Binarized Feature Matching', 'gao24b_interspeech', 'bfm method aec webrtc dearth nn-based traditional encompassing canceller serf'], ['Yang Ai, Ye-Xin Lu, Xiao-Hang Jiang, Zheng-Yan Sheng, Rui-Chen Zheng, Zhen-Hua Ling', 'A Low-Bitrate Neural Audio Codec Framework with Bandwidth Reduction and Recovery for High-Sampling-Rate Waveforms', 'ai24b_interspeech', 'bitrate waveform quantizer sampling encoder decoder outputted bitrates decodes recovers'], ['Toshio Irino, Shintaro Doan, Minami Ishikawa', 'Signal processing algorithm effective for sound quality of hearing loss simulators ', 'irino24_interspeech', 'whis perceptible simulator distortion less filterbank version listener synthesis cambridge'], ['Hualei Wang, Jianguo Mao, Zhifang Guo, Jiarui Wan, Hong Liu, Xiangdong Wang', 'Leveraging Language Model Capabilities for Sound Event Detection', 'wang24d_interspeech', 'multi-modality audio generation timestamps sed feature showcase abundant classification flexibly'], ['Mohammad Hassan Vali, Tom Bäckström', 'Privacy PORCUPINE: Anonymization of Speaker Attributes Using Occurrence Normalization for Space-Filling Vector Quantization', 'vali24_interspeech', 'information bottleneck private thus disclosure quantifiable privacy-preserving uneven protecting quantizing'], ['Jingyao Wu, Ting Dang, Vidhyasaharan Sethu, Eliathamby Ambikairajah', 'Dual-Constrained Dynamical Neural ODEs for Ambiguity-aware Continuous Emotion Prediction', 'wu24_interspeech', 'distribution ambiguity recola parameterised dynamic range evolve restrict smoothly less'], ['Suhas BN, Amanda Rebar, Saeed Abdullah', 'Speaking of Health: Leveraging Large Language Models to assess Exercise Motivation and Behavior of Rehabilitation Patients', 'bn24_interspeech', 'session establish outcome wellbeing compliance decision-making thematic sentiment ground-truth augmenting'], ['Yu-Wen Chen, Zhou Yu, Julia Hirschberg', 'MultiPA: A Multi-task Speech Pronunciation Assessment Model for Open Response Scenarios', 'chen24c_interspeech', 'sentence-level open-response accuracy multitask offering predominantly real-life out-of-domain reached fluency'], ['Candy Olivia Mawalim, Shogo Okada, Masashi Unoki', 'Are Recent Deep Learning-Based Speech Enhancement Methods Ready to Confront Real-World Noisy Environments?', 'mawalim24_interspeech', 'fullsubnet denoiser scenario datasets hallucination nuanced assess asr-based emphasizes struggle'], ['Jihyun Mun, Sunhee Kim, Minhwa Chung', 'Developing an End-to-End Framework for Predicting the Social Communication Severity Scores of Children with Autism Spectrum Disorder', 'mun24_interspeech', 'asd diagnostic fine-tuned human-rated lifelong tool necessitate foundational profound objective'], ['Mewlude Nijat, Chen Chen, Dong Wang, Askar Hamdulla', 'UY/CH-CHILD -- A Public Chinese L2 Speech Database of Uyghur Children', 'nijat24_interspeech', 'kindergarten pronunciation school primary intriguing mature downloaded avenue showcase org'], ['Steven Vander Eeckt, Hugo Van hamme', 'Unsupervised Online Continual Learning for Automatic Speech Recognition', 'vandereeckt24_interspeech', 'ocl uocl forgetting continually leveraging asr domain supervised adaptation realm'], ['Robin Scheibler, Yusuke Fujita, Yuma Shirahata, Tatsuya Komatsu', 'Universal Score-based Speech Enhancement with High Content Preservation', 'scheibler24_interspeech', 'universe diffusion adversarial three-fold loss low-rank favorably fidelity promote stability'], ['Minyoung Lee, Eunil Park, Sungeun Hong', 'FVTTS : Face Based Voice Synthesis for Text-to-Speech', 'lee24_interspeech', 'face-based personalized expressive identity image tt sample individual voice-based personalization'], ['Liang Tao, Maoshen Jia, Yonggang Hu, Changchun Bao', 'Spatial Acoustic Enhancement Using Unbiased Relative Harmonic Coefficients', 'tao24_interspeech', 'n
11146oisy source entire spherical multiplying compactly denoted environment domain clue'], ['Xin Jing, Luyang Zhang, Jiangjian Xie, Alexander Gebhard, Alice Baird, Björn Schuller', 'DB3V: A Dialect Dominated Dataset of Bird Vocalisation for Cross-corpus Bird Species Recognition', 'jing24_interspeech', 'publicly call zenodo region impedes contiguous mitigation benchmarking across acknowledged'], ['Da Mu, Zhicheng Zhang, Haobo Yue', 'MFF-EINV2: Multi-scale Feature Fusion across Spectral-Spatial-Temporal Domains for Sound Event Localization and Detection', 'mu24_interspeech', 'mff seld spatial module temporal spectral subnetworks localizing three-stage challenge'], ['Hao Yang, Min Zhang, Minghan Wang, Jiaxin Guo', 'RASU: Retrieval Augmented Speech Understanding through Generative Modeling', 'yang24b_interspeech', 'slu prompt retrieved rag language intent transcript spoken capability retrieval-based'], ['Julius Richter, Yi-Chiao Wu, Steven Krenn, Simon Welker, Bunlong Lay, Shinji Watanabe, Alexander Richard, Timo Gerkmann', 'EARS: An Anechoic Fullband Speech Dataset Benchmarked for Speech Enhancement and Dereverberation', 'richter24_interspeech', 'online freeform totalling style uploaded download non-verbal instrumental server blind'], ['Vladimir Despotovic, Abir Elbéji, Petr V. Nazarov, Guy Fagherazzi', 'Multimodal Fusion for Vocal Biomarkers Using Vector Cross-Attention', 'despotovic24_interspeech', 'modality voice standardized sustained phonation person reading biomarker single vowel'], ['Liming Wang, Yuan Gong, Nauman Dawalatabad, Marco Vilela, Katerina Placek, Brian Tracey, Yishu Gong, Alan Premasiri, Fernando Vieira, James Glass', 'Automatic Prediction of Amyotrophic Lateral Sclerosis Progression using Longitudinal Speech Transformer', 'wang24e_interspeech', 'al disease recording best auc interpretable network-based careful fine-grained pretrained'], ['Yujia Wang, Hexin Liu, Leibny Paola Garcia', 'Bridging Child-Centered Speech Language Identification and Language Diarization via Phonetics', 'wang24f_interspeech', 'lid fine-tuning back-end scoring pre-trained merlion slicing enriching inspiration needing'], ['Erfan Loweimi, Mengjie Qian, Kate Knill, Mark Gales', 'On the Usefulness of Speaker Embeddings for Speaker Retrieval in the Wild: A Comparative Study of x-vector and ECAPA-TDNN Models', 'loweimi24_interspeech', 'synopsis speechbrain develop nemo bbc archival posing assess toolkits encompasses'], ['Zhengxiao Li, Nakamasa Inoue', 'Locally Aligned Rectified Flow Model for Speech Enhancement Towards Single-Step Diffusion', 'li24d_interspeech', 'larf transport equation differential wsj mapping voicebank-demand clean si-sdr noisy'], ['Jeehye Lee, Hyeji Seo', 'Online Knowledge Distillation of Decoder-Only Large Language Models for Efficient Speech Recognition', 'lee24b_interspeech', 'llm cer inference decoder reduction relative cost aed dataset task'], ['Yixuan Zhang, Hao Zhang, Meng Yu, Dong Yu', 'Neural Network Augmented Kalman Filter for Robust Acoustic Howling Suppression', 'zhang24_interspeech', 'ahs refining standalone adaptability performance method covariance validate thereby obtaining'], ['Dong Yang, Tomoki Koriyama, Yuki Saito', 'Frame-Wise Breath Detection with Self-Training: An Exploration of Enhancing Breath Naturalness in Text-to-Speech', 'yang24c_interspeech', 'annotation model mark up-sampling tt pseudo-labeling position synthesizes human-like conformer'], ['Hitoshi Suda, Aya Watanabe, Shinnosuke Takamichi', 'Who Finds This Voice Attractive? A Large-Scale Experiment Using In-the-Wild Data', 'suda24_interspeech', 'likability gender regarding age favorite corpus speech announcement listener speaker'], ['Mayank Kumar Singh, Naoya Takahashi, Weihsiang Liao, Yuki Mitsufuji', 'SilentCipher: Deep Audio Watermarking', 'singh24_interspeech', 'message imperceptible robustness introduce learning-based encode capacity enhancing enabling watermark'], ['Hyun Myung Kim, Kangwook Jang, Hoirin Kim', 'One-class learning with adaptive centroid shift for audio deepfake detection', 'kim24b_interspeech', 'bonafide ac representation unseen embeddings cluster well-separated t-sne disentangles system'], ['Constantijn Kaland, Jeremy Steffman, Jennifer Cole', 'K-means and hierarchical clustering of f0 contours', 'kaland24_interspeech', 'cluster intonation outcome difference popular research decision analysis time-series imitated'], ['Constantijn Kaland, Maria Lialiou', 'Quantity-sensitivity affects recall performance of word stress', 'kaland24b_interspeech', 'var pattern listener greek language segmental memorizing german variable hypothesizes'], ['Loukas Ilias, Dimitris Askounis', 'A Cross-Attention Layer coupled with Multimodal Fusion Methods for Recognizing Depression from Spontaneous Speech', 'ilias24_interspeech', 'perform lifeless activity people biomarker existing diagnosing feel depressed productivity'], ['Francesco Paissan, Elisabetta Farella', 'tinyCLAP: Distilling Constrastive Language-Audio Pretrained Models', 'paissan24_interspeech', 'clap contrastive event complexity text-to-audio detection employment sound less distillation'], ['Yiwen Wang, Xihong Wu', 'TSE-PI: Target Sound Extraction under Reverberant Environments with Pitch Information', 'wang24g_interspeech', 'tse gammatone filterbank fsd feature-wise provided asa model learnable clue'], ['Chin Yuen Kwok, Jia Qi Yip, Eng Siong Chng', 'Continual Learning Optimizations for Auto-regressive Decoder of Multilingual ASR systems', 'kwok24_interspeech', 'masr token pre-trained re-scaling unused freezing hypothesise surgery language sub-optimal'], ['Nicolas Gengembre, Olivier Le Blouch, Cédric Gendrot', 'Disentangling prosody and timbre embeddings via voice conversion', 'gengembre24_interspeech', 'converted anonymization architecture speaker perfectly expressivity source prosodic disentangle preservation'], ['Masato Murata, Koichi Miyazaki, Tomoki Koriyama', 'An Attribute Interpolation Method in Speech Synthesis by Model Merging', 'murata24_interspeech', 'base merged task intensity emotion generation module control specific creates'], ['Quan Wang, Yiling Huang, Guanlong Zhao, Evan Clark, Wei Xia, Hank Liao', 'DiarizationLM: Speaker Diarization Post-Processing with Large Language Models', 'wang24h_interspeech', 'llm rel finetuned output framework wder post-process optionally dataset off-the-shelf'], ['Pawel Bujnowski, Bartlomiej Kuzma, Bartlomiej Paziewski, Jacek Rutkowski, Joanna Marhula, Zuzanna Bordzicka, Piotr Andruszkiewicz', 'SAMSEMO: New dataset for multilingual and multimodal emotion recognition', 'bujnowski24_interspeech', 'scene video cmu-mosei datasets class imbalanced audio metadata popularity heterogeneous'], ['Hyeonuk Nam, Seong-Hu Kim, Deokki Min, Junhyeok Lee, Yong-Hwa Park', 'Diversifying and Expanding Frequency-Adaptive Convolution Kernels for Sound Event Detection', 'nam24_interspeech', 'conv basis dilation dilated size frequency dfd diversify class-wise psds'], ['Jihwan Lee, Aditya Kommineni, Tiantian Feng, Kleanthis Avramidis, Xuan Shi, Sudarsana Reddy Kadiri, Shrikanth Narayanan', 'Toward Fully-End-to-End Listened Speech Decoding from EEG Signals', 'lee24c_interspeech', 'connector module learns waveform single-step unveil characteristic framework bridge com'], ['Tuochao Chen, Qirui Wang, Bohan Wu, Malek Itani, Emre Sef
11146ik Eskimez, Takuya Yoshioka, Shyamnath Gollakota', 'Target conversation extraction: Source separation using turn-taking dynamics', 'chen24d_interspeech', 'interfering -speaker speaker amidst participant uniquely accomplish engaged noise feasibility'], ['Youngmoon Jung, Seungjin Lee, Joon-Young Yang, Jaeyoung Roh, Chang Woo Han, Hoon-Young Cho', 'Relational Proxy Loss for Audio-Text based Keyword Spotting', 'jung24_interspeech', 'embeddings text kw enrollment within proxy-based point-to-point acoustic convenience focus'], ['Jaejun Lee, Yoori Oh, Injune Hwang, Kyogu Lee', 'Hear Your Face: Face-based voice conversion with F0 estimation', 'lee24d_interspeech', 'facial delf fundamental characteristic target framework frequency emerging align solely'], ['Eunik Park, Daehyun Ahn, Hyungjun Kim', 'RepTor: Re-parameterizable Temporal Convolution for Keyword Spotting via Differentiable Kernel Search', 'park24_interspeech', 'kw device mobile model re-parameterized galaxy top- plain smart feed-forward'], ['Seonwoo Lee, Sunhee Kim, Minhwa Chung', 'Automatic Assessment of Speech Production Skills for Children with Cochlear Implants Using Wav2Vec2.0 Acoustic Embeddings', 'lee24e_interspeech', 'ci adult model korean-speaking multi-head pearson therapy trait assessing self-supervised'], ['Hokuto Munakata, Ryo Terashima, Yusuke Fujita', 'Song Data Cleansing for End-to-End Neural Singer Diarization Using Neural Analysis and Synthesis Framework', 'munakata24_interspeech', 'singing choral eend non-overlapped solo convert popular duet mitigates model'], ['Heinrich Dinkel, Zhiyong Yan, Yongqing Wang, Junbo Zhang, Yujun Wang, Bin Wang', 'Streaming Audio Transformers for Online Audio Tagging', 'dinkel24_interspeech', 'sat sota usage delay memory vit benchmarked audioset widely-used long-range'], ['Qinglin Meng, Min Liu, Kaixun Huang, Kun Wei, Lei Xie, Zongfeng Quan, Weihong Deng, Quan Lu, Ning Jiang, Guoqing Zhao', 'SEQ-former: A context-enhanced and efficient automatic speech recognition framework', 'meng24_interspeech', 'contextual efficiency ctc asr decoder spike reduce prediction information private'], ['Yusuke Fujita, Tatsuya Komatsu', 'Audio Fingerprinting with Holographic Reduced Representations', 'fujita24_interspeech', 'fingerprint resolution time decimation method combined original circular summation aggregation'], ['Heinrich Dinkel, Zhiyong Yan, Yongqing Wang, Junbo Zhang, Yujun Wang, Bin Wang', 'Scaling up masked audio encoder learning for general audio classification', 'dinkel24b_interspeech', 'music environmental ssl voxlingua competes nearest-neighbor crema-d ssl-based speech billion'], ['Nicolas M. Müller, Nicholas Evans, Hemlata Tak, Philip Sperl, Konstantin Böttinger', 'Harder or Different? Understanding Generalization of Audio Deepfake Detection', 'muller24b_interspeech', 'deepfakes hardness component gap model newer question generated fundamentally decomposing'], ['Shihao Chen, Yu Gu, Jie Zhang, Na Li, Rilin Chen, Liping Chen, Lirong Dai', 'LDM-SVC: Latent Diffusion Model Based Zero-Shot Any-to-Any Singing Voice Conversion with Singer Guidance', 'chen24e_interspeech', 'svc timbre ldm classifier-free pretrain original vits leakage inevitable editing'], ['Haochen Wu, Wu Guo, Zhentao Zhang, Wenting Zhao, Shengyu Peng, Jie Zhang', 'Spoofing Speech Detection by Modeling Local Spectro-Temporal and Long-term Dependency', 'wu24b_interspeech', 'ssd branch artifact temporal bilstm-based exploit global dual-branch attention reside'], ['Haochen Wu, Wu Guo, Shengyu Peng, Zhuhai Li, Jie Zhang', 'Adapter Learning from Pre-trained Model for Robust Spoof Speech Detection', 'wu24c_interspeech', 'backbone wav vec local-global freezing appended task-dependent catastrophic over-fitting wavlm'], ['Yuankun Xie, Ruibo Fu, Zhengqi Wen, Zhiyong Wang, Xiaopeng Wang, Haonnan Cheng, Long Ye, Jianhua Tao', 'Generalized Source Tracing: Detecting Novel Audio Deepfake Algorithm with Real Emphasis and Fake Dispersion Strategy', 'xie24_interspeech', 'ood identifying nsd post-hoc showcasing logits sample urgent proliferation detection'], ['Hui-Peng Du, Ye-Xin Lu, Yang Ai, Zhen-Hua Ling', 'BiVocoder: A Bidirectional Neural Vocoder Integrating Feature Extraction and Waveform Generation', 'du24_interspeech', 'stft amplitude phase tt restores spectrum comprehensively symmetric analysis-synthesis reverse'], ['Tomoki Honda, Shinsuke Sakai, Tatsuya Kawahara', 'Efficient and Robust Long-Form Speech Recognition with Hybrid H3-Conformer', 'honda24_interspeech', 'mhsa hungry layer self-attention computation state-space csj multi-head deterioration unreliable'], ['Shuaishuai Ye, Shunfei Chen, Xinhui Hu, Xinkang Xu', 'SC-MoE: Switch Conformer Mixture of Experts for Unified Streaming and Non-streaming Code-Switching ASR', 'ye24_interspeech', 'moe router layer encoder decoder blank language equipped lid ctc'], ['Shengyu Peng, Wu Guo, Haochen Wu, Zuoliang Li, Jie Zhang', 'Fine-tune Pre-Trained Models with Multi-Level Feature Fusion for Speaker Verification', 'peng24_interspeech', 'ptm afm dual-branch merge back-end front-end ptms layer fbank aggregated'], ['Tongtao Ling, Yutao Lai, Lei Chen, Shilei Huang, Yi Liu', 'A Small and Fast BERT for Chinese Medical Punctuation Restoration', 'ling24_interspeech', 'pre-training apr fine-tuning mark clinical pre-trained model roberta reformulate distilled'], ['Qian Yang, Jialong Zuo, Zhe Su, Ziyue Jiang, Mingze Li, Zhou Zhao, Feiyang Chen, Zhefeng Wang, Baoxing Huai', 'MSceneSpeech: A Multi-Scene Speech Dataset For Expressive Speech Synthesis', 'yang24d_interspeech', 'open style scenario user-specific prosody prompting multiple source entail audio'], ['Malo Maisonneuve, Corinne Fredouille, Muriel Lalain, Alain Ghio, Virginie Woisard', 'Towards objective and interpretable speech disorder assessment: a comparative analysis of CNN and transformer-based models', 'maisonneuve24_interspeech', 'hnc pathological wav vec patient impact datasets neck unbiased pave'], ['Ryo 
11146Setoguchi, Yoshiko Arimoto', 'Acoustical analysis of the initial phones in speech-laugh', 'setoguchi24_interspeech', 'anova speech vowel glmms th-order lower-order variable coefficient explanatory elucidate'], ['Hao Shi, Tatsuya Kawahara', 'Dual-path Adaptation of Pretrained Feature Extraction Module for Robust Automatic Speech Recognition', 'shi24b_interspeech', 'adapter finetuning path encoder noisy effective transformer achieving data freezing'], ['Thanh Lan Truong, Andrea Weber', 'Ethnolinguistic Identification of Vietnamese-German Heritage Speech', 'truong24_interspeech', 'asian german vietnamese listener speaker background identify chance accurately monotonized'], ['Oliver Niebuhr, Nafiseh Taghva', 'How rhythm metrics are linked to produced and perceived speaker charisma', 'niebuhr24_interspeech', 'presentation investor rhythmic element charismatic medium-sized duration-based committed recorded pvi'], ['Shahin Amiriparian, Filip Packań, Maurice Gerczuk, Björn W. Schuller', 'ExHuBERT: Enhancing HuBERT Through Block Extension and Fine-Tuning on 37 Emotion Datasets', 'amiriparian24_interspeech', 'ser duplicate freeze huggingface layer skip gather adaptability twofold backbone'], ['Jie Lin, Xiuping Yang, Li Xiao, Xinhong Li, Weiyan Yi, Yuhong Yang, Weiping Tu, Xiong Chen', 'SimuSOE: A Simulated Snoring Dataset for Obstructive Sleep Apnea-Hypopnea Syndrome Evaluation during Wakefulness', 'lin24_interspeech', 'osahs snore obstruction airway upper signal datasets intentionally emitted chronic'], ['Shoval Messica, Yossi Adi', 'NAST: Noise Aware Speech Tokenization for Speech Language Models', 'messica24_interspeech', 'setup task disentanglement lastly representation com github serf iii signal'], ['Chiara Riegger, Tina Bögel, George Walkden', 'The prosody of the verbal prefix ge-: historical and experimental evidence', 'riegger24_interspeech', 'old negation preceding trochaic foot closure rhythmic auxiliary principle word'], ['Kexu Liu, Yuanxin Wang, Shengchen Li, Xi Shao', 'Speech Formants Integration for Generalized Detection of Synthetic Speech Spoofing Attacks', 'liu24b_interspeech', 'multi-view xls-r unseen variance feature one-class compactly dynamic gating struggle'], ['Zhe Li, Man-wai Mak, Hung-yi Lee, Helen Meng', 'Parameter-efficient Fine-tuning of Speaker-Aware Dynamic Prompts for Speaker Verification', 'li24e_interspeech', 'pool prompt transformer pre-trained tuning tunable resulting overfit plasticity learnable'], ['Bolaji Yusuf, Murali Karthick Baskar, Andrew Rosenberg, Bhuvana Ramabhadran', 'Speculative Speech Recognition by Audio-Prefixed Low-Rank Adaptation of Language Models', 'yusuf24_interspeech', 'ssr asr speculates empower speculation transcribes ahead prefix audio feed'], ['Louis Abel, Vincent Colotte, Slim Ouni', 'Towards realtime co-speech gestures synthesis using STARGATE', 'abel24_interspeech', 'graph fast ecas autoregression field credibility resource-intensive audio-text spatio-temporal architecture'], ['Chenyu Li, Jalal Al-Tamimi', 'Impact of the tonal factor on diphthong realizations in Standard Mandarin with Generalized Additive Mixed Models', 'li24f_interspeech', 'tone falling tend realization gamms universality monophthongs vowel negatively dynamical'], ['Yehoshua Dissen, Shiry Yonash, Israel Cohen, Joseph Keshet', 'Enhanced ASR Robustness to Packet Loss with a Front-End Adaptation Network', 'dissen24_interspeech', 'whisper model packet-loss criterion practicality underscoring foundational noisy realm environment'], ['Nicolas Audibert, Cecile Fougeron, Christine Meunier', 'Do Speaker-dependent Vowel Characteristics depend on Speech Style?', 'audibert24_interspeech', 'speaker intra-speaker nasal distinction style-dependent read category variability spontaneous property'], ['Riyansha Singh, Parinita Nema, Vinod K Kurmi', 'Towards Robust Few-shot Class Incremental Learning in Audio Classification using Contrastive Representation', 'singh24b_interspeech', 'session base esc danger post-training analytics seamless address catastrophic forgetting'], ['Amit Roth, Arnon Turetzky, Yossi Adi', 'A Language Modeling Approach to Diacritic-Free Hebrew TTS', 'roth24_interspeech', 'diacritic modern word-piece text-to-speech tokenizer dictate in-the-wild imposes weakly pronounce'], ['Veranika Boukun, Jakob Drefs, Jörg Lücke', 'Blind Zero-Shot Audio Restoration: A V
11146ariational Autoencoder Approach for Denoising and Inpainting', 'boukun24_interspeech', 'optimization signal posterior probabilistic setting challenging truncated available grounded theoretically'], ['Austin Jones, Margaret E. L. Renwick', 'Evaluating Italian Vowel Variation with the Recurrent Neural Network Phonet', 'jones24_interspeech', 'posterior phonological lexicon close label phone correlation sociophonetic class marginally'], ['Harsha Veena Tadavarthy, Austin Jones, Margaret E. L. Renwick', 'Phonological Feature Detection for US English using the Phonet Library', 'tadavarthy24_interspeech', 'probability class distinctive posterior relationship patterned timestamps bridging designated broader'], ['Andrew Rouditchenko, Yuan Gong, Samuel Thomas, Leonid Karlinsky, Hilde Kuehne, Rogerio Feris, James Glass', 'Whisper-Flamingo: Integrating Visual Features into Whisper for Audio-Visual Speech Recognition and Translation', 'rouditchenko24_interspeech', 'video avsr thousand model hour language versatile data gated harder'], ['Soham Deshmukh, Dareen Alharthi, Benjamin Elizalde, Hannes Gamper, Mahmoud Al Ismail, Rita Singh, Bhiksha Raj, Huaming Wang', 'PAM: Prompting Audio-Language Models for Audio Quality Assessment', 'deshmukh24b_interspeech', 'alm metric reference-free score listening human prompt computing ttm tta'], ['Yi-Jen Shih, David Harwath', 'Interface Design for Self-Supervised Speech Models', 'shih24_interspeech', 'ssl downstream upstream depth sum many weighted layerwise logarithmically combining'], ['Helin Wang, Jesús Villalba, Laureano Moro-Velazquez, Jiarui Hai, Thomas Thebaud, Najim Dehak', 'Noise-robust Speech Separation with Fast Generative Correction', 'wang24i_interspeech', 'diffusion reverse process corrector rectify streamline sepformer si-snr libri isolating'], ['Suhita Ghosh, Melanie Jouaiti, Arnab Das, Yamini Sinha, Tim Polzehl, Ingo Siegert, Sebastian Stober', 'Anonymising Elderly and Pathological Speech: Voice Conversion Using DDSP and Query-by-Example', 'ghosh24_interspeech', 'domain preservation anonymity anonymisation uncommon prosody clinically disentangling pertinent differentiable'], ['Kunal Dhawan, Nithin Rao Koluguri, Ante Jukić, Ryan Langman, Jagadeesh Balam, Boris Ginsburg', 'Codec-ASR: Training Performant Automatic Speech Recognition Systems with Discrete Speech Representations', 'dhawan24_interspeech', 'asr c
11146odec ml-superb encodec garnered foundational speech-text speech-related bit-rate less'], ['Vidya Srinivas, Malek Itani, Tuochao Chen, Emre Sefik Eskimez, Takuya Yoshioka, Shyamnath Gollakota', 'Knowledge boosting during low-latency inference', 'srinivas24_interspeech', 'model chunk streaming operate small running delay large larger time-delayed'], ['Kyuhong Shim, Jinkyu Lee, Hyunjae Kim', 'Leveraging Adapter for Parameter-Efficient ASR Encoder', 'shim24_interspeech', 'parameter parameter-sharing reduces reuses module adjusts conformer-based architecture insert balancing'], ['Paarth Neekhara, Shehzeen Hussain, Subhankar Ghosh, Jason Li, Boris Ginsburg', 'Improving Robustness of LLM-based Speech Synthesis by Learning Monotonic Alignment', 'neekhara24_interspeech', 'cross-attention token text tt attention model hallucination robust repeating learnable'], ['Hongmei Guo, Yijiang Chen, Xiao-Lei Zhang, Xuelong Li', 'Graph Attention Based Multi-Channel U-Net for Speech Dereverberation With Ad-Hoc Microphone Arrays', 'guo24_interspeech', 'channel module reverberation model fusion integrated studied selection layer train'], ['Jia Qi Yip, Shengkui Zhao, Dianwen Ng, Eng Siong Chng, Bin Ma', 'Towards Audio Codec-based Speech Separation', 'yip24_interspeech', 'nac compression task sepformer mac chart cloud impractical codecs adopting'], ['Nigel G. Ward, Andres Segura, Alejandro Ceballos, Divette Marco', 'Towards a General-Purpose Model of Perceived Pragmatic Similarity', 'ward24_interspeech', 'human inter-annotator generality hubert utterance fairly thousand judge sometimes judgment'], ['Yuma Shirahata, Byeongseon Park, Ryuichi Yamamoto, Kentaro Tachibana', 'Audio-conditioned phonemic and prosodic annotation for building text-to-speech models from unlabeled speech data', 'shirahata24_interspeech', 'label-speech tt paired dataset model trained sample existing shortage text-only'], ['Jingjing Xu, Wei Zhou, Zijian Yang, Eugen Beck, Ralf Schlüter', 'Dynamic Encoder Size Based on Data-Driven Layer-wise Pruning for Speech Recognition', 'xu24_interspeech', 'supernet subnets performant on-par full-size effort score-based resource-intensive enjoy model'], ['Zhuhai Li, Jie Zhang, Wu Guo, Haochen Wu', 'Boosting the Transferability of Adversarial Examples with Gradient-Aligned Ensemble Attack for Speaker Recognition', 'li24g_interspeech', 'substitute gradient update model victim black-box spoof randomly calculate voxceleb'], ['Run Chen, Haozhe Chen, Anushka Kulkarni, Eleanor Lin, Linda Pang, Divya Tadimeti, Jun Shin, Julia Hirschberg', 'Detecting Empathy in Speech', 'chen24f_interspeech', 'feeling acoustic-prosodic done empathetic likability trust benchmarking interpretable creating identifying'], ['Yerbolat Khassanov, Zhipeng Chen, Tianfeng Chen, Tze Yuang Chong, Wei Li, Jun Zhang, Lu Lu, Yuxuan Wang', 'Dual-Pipeline with Low-Rank Adaptation for New Language Integration in Multilingual ASR', 'khassanov24_interspeech', 'masr pipeline lora pre-trained existing flow decoder fleurs language-agnostic unavailable'], ['Kohei Matsuura, Takanori Ashihara, Takafumi Moriya, Masato Mimura, Takatomo Kano, Atsunori Ogawa, Marc Delcroix', 'Sentence-wise Speech Summarization: Task, Datasets, and End-to-End Modeling with LM Knowledge Distillation', 'matsuura24_interspeech', 'cascade model summary text sentence-by-sentence combine appealing asr transformer-based evaluates'], ['Zezhong Jin, Youzhi Tu, Man-Wai Mak', 'W-GVKT: Within-Global-View Knowledge Transfer for Speaker Verification', 'jin24b_interspeech', 'dino student view global teacher eer diversification non-contrastive diversified negligible'], ['Michael Lambropoulos, Frantz Clermont, Shunichi Ishihara', 'The sub-band cepstrum as a tool for locating local spectral regions of phonetic sensitivity: A first attempt with multi-speaker vowel data', 'lambropoulos24_interspeech', 'blccs sub-bands flexible cc spectrum band-limited implying full-band classification gaining'], ['Kun Zhou, Shengkui Zhao, Yukun Ma, Chong Zhang, Hao Wang, Dianwen Ng, Chongjia Ni, Trung Hieu Nguyen, Jia Qi Yip, Bin Ma', 'Phonetic Enhanced Language Modeling for Text-to-Speech Synthesis', 'zhou24_interspeech', 'autoregressive non-autoregressive tt in-context training accumulation model scalability codecs propagation'], ['Zezhong Jin, Youzhi Tu, Man-Wai Mak', 'Self-Supervised Learning with Multi-Head Multi-Mode Knowledge Distillation for Speaker Verification', 'jin24c_interspeech', 'memo self architecture teacher contrastive head mode student dino supportive'], ['Ju-ho Kim, Hee-Soo Heo, Bong-Jin Lee, Youngki Kwon, Minjae Lee, Ha-Jin Yu', 'Self-supervised speaker verification with relational mask prediction', 'kim24c_interspeech', 'ssl-based overlooking inter-frame comprehensively mitigating encourages aggregation enrich bridge emerged'], ['Eungbeom Kim, Hantae Kim, Kyogu Lee', 'Guiding Frame-Level CTC Alignments Using Self-knowledge Distillation', 'kim24d_interspeech', 'disagreement student alignment encoder introduces sub-model method spike model teacher-student'], ['Yao Shen, Yingying Gao, Yaqian Hao, Chenguang Hu, Fulin Zhang, Junlan Feng, Shilei Zhang', 'CEC: A Noisy Label Detection Method for Speaker Recognition', 'shen24_interspeech', 'counting inconsistent sample hard cic overfitted metric excels preventing categorize'], ['En-Lun Yu, Kuan-Hsun Ho, Jeih-weih Hung, Shih-Chieh Huang, Berlin Chen', 'Speaker Conditional Sinc-Extractor for Personal VAD', 'yu24_interspeech', 'pvad vanilla feature sinc d-vectors wearable cutoff acoustic accepting function'], ['Seyun Um, Doyeon Kim, Hong-Goo Kang', 'PARAN: Variational Autoencoder-based End-to-End Articulation-to-Speech System for Speech Intelligibility', 'um24_interspeech', 'ema latent signal adjusts distribution researched vae high-fidelity normalizing clarity'], ['Jingze Lu, Yuxiang Zhang, Zhuo Li, Zengqiang Shang, Wenchao Wang, Pengyuan Zhang', 'Improving Copy-Synthesis Anti-Spoofing Training Method with Rhythm and Speaker Perturbation', 'lu24b_interspeech', 'artifact algorithm tt introduced neglecting acoustic model facing locate vocoders'], ['Sheng Li, Chen Chen, Chin Yuen Kwok, Chenhui Chu, Eng Siong Chng, Hisashi Kawai', 'Investigating ASR Error Correction with Large Language Model and Multilingual 1-best Hypotheses', 'li24h_interspeech', 'llm n-best effectively correct llm-based low-resourced feeding noticed output let'], ['Yip Keng Kan, Ke Xu, Hao Li, Jie Shi', 'VoiceDefense: Protecting Automatic Speaker Verification Models Against Black-box Adversarial Attacks', 'kan24_interspeech', 'asv sample trustworthiness formidable counteract slice detection proving compromised distinctly'], ['HyunJung Choi, Muyeol Choi, Yohan Lim, Minkyu Lee, Seonhui Kim, Seung Yun, Donghyun Kim, SangHun Kim', 'Spoken-to-written text conversion with Large Language Model', 'choi24_interspeech', 'itn readability notation korean written making standardize pronunciation err form'], ['Jiwon Suh, Injae Na, Woohwan Jung', 'Improving Domain-Specific ASR with LLM-Generated Contextual Descriptions', 'suh24_interspeech', 'datasets terminology specific domain struggle metadata unavailable llm method whisper'], ['Rubing Shen, Yanzhen Ren, Zongkun Sun', 'FA-GAN: Artifacts-free and Phase-aware High-fidelity GAN-based Vocoder', 'shen24b_interspeech', 'artifact spectral quality non-ideal blurring twin alleviating aliasing upsampling deconvolution'], ['Sheng Feng, Heyang Liu, Yu Wang, Yanfeng Wang', 'Towards an End-to-End Framework for Invasive Brain Signal Decoding with Large Language Models', 'feng24_interspeech', 'llm groundbreaking bcis bci brain-computer immense underscoring evolve underscore showcase'], ['Yi-Wei Wang, Ke-Han Lu, Kuan-Yu Chen', 'HypR: A comprehensive study for ASR hypothesis revising with a reference corpus', 'wang24j_interspeech', 'recognition error progress research checkpoint reranking ted-lium speech dataset modeling'], ['Yuki Saito, Takuto Igarashi, Kentaro Seki, Shinnosuke Takamichi, Ryuichi Yamamoto, Kentaro Tachibana, Hiroshi Saruwatari', 'SRC4VC: Smartphone-Recorded Corpus for Voice Conversion Benchmark', 'saito24_interspeech', 'sample low-quality smartphones multi-speaker high-quality speech speaker-wise utterance-wise any-to-any recorded'], ['Jonathan Svirsky, Uri Shaham, Ofir Lindenbaum', 'Sparse Binarization for Fast Keyword Spotting', 'svirsky24_interspeech', 'device kw edge model efficiency keyword-spotting voice-activated necessitates convenience smartphones'], ['Aviv Shamsian, Aviv Navon, Neta Glazer, Gill Hetz, Joseph Keshet', 'Keyword-Guided Adaptation of Automatic Speech Recognition', 'shamsian24_interspeech', 'whisper prompt decoder whisper-based steer transcription biasing prefix guiding significant'], ['Guillem Bonafos, Clara Bourot, Pierre Pudlo, Jean-Marc Freyermuth, Laurence Reboul, Samuel Tronçon, Arnaud Rey', 'Dirichlet process mixture model based on topologically augmented signal representation for clustering infant vocalizations', 'bonafos24_interspeech', 'month diagram vocalization life persistence persistent non-parametric categorization mfccs mel-frequency'], ['Asad Ullah, Alessandro Ragano, Andrew Hines', 'Reduce, Reuse, Recycle: Is Perturbed Data Better than Other Language Augmentation for Low Resource Self
11146-Supervised Speech Models', 'ullah24_interspeech', 'pre-training combined resource-constrained phoneme pre-train pitch target noise viable option'], ['Deok-Hyeon Cho, Hyung-Seok Oh, Seung-Bin Kim, Sang-Hoon Lee, Seong-Whan Lee', 'EmoSphere-TTS: Emotional Style and Intensity Modeling via Spherical Emotion Vector for Controllable Emotional Text-to-Speech', 'cho24_interspeech', 'ability expressive speech multi-aspect control pseudo-labels nuanced mimicking synthesizes compromising'], ['Chenyuan Zhang, Linkai Luo, Hong Peng, Wei Wen', 'Variable Segment Length and Domain-Adapted Feature Optimization for Speaker Diarization', 'zhang24b_interspeech', 'msr mixed embeddings still multiple alternation distinguishes surpasses unreliable reaching'], ['Marie Kunešová, Jan Lehečka, Josef Michálek, Jindrich Matousek, Jan Švec', 'Zero-shot Out-of-domain is No Joke: Lessons Learned in the VoiceMOS 2023 MOS Prediction Challenge', 'kunesova24_interspeech', 'track ensemble win surpassed centered singing degrade wav team vec'], ['Eros Rosello, Angel M. Gomez, Iván López-Espejo, Antonio M. Peinado, Juan M. Martín-Doñas', 'Anti-spoofing Ensembling Model: Dynamic Weight Allocation in Ensemble Models for Improved Voice Biometrics Security', 'rosello24_interspeech', 'countermeasure showcasing still neglecting malicious adjusts susceptible spoofed neural prof'], ['Tianzi Wang, Xurong Xie, Zhaoqing Li, Shoukang Hu, Zengrui Jin, Jiajun Deng, Mingyu Cui, Shujie Hu, Mengzhe Geng, Guinan Li, Helen Meng, Xunying Liu', 'Towards Effective and Efficient Non-autoregressive Decoding Using Block-based Attention Mask', 'wang24k_interspeech', 'amd ctc nar decoder block statistically librispeech- concealed wer tripartite'], ['Yifei Xin, Xuxin Cheng, Zhihong Zhu, Xusheng Yang, Yuexian Zou', 'DiffATR: Diffusion-based Generative Modeling for Audio-Text Retrieval', 'xin24_interspeech', 'atr query candidate methodology joint discriminative clotho audiocaps loss discerning'], ['Hiroshi Sato, Takafumi Moriya, Masato Mimura, Shota Horiguchi, Tsubasa Ochiai, Takanori Ashihara, Atsushi Ando, Kentaro Shinayama, Marc Delcroix', 'SpeakerBeam-SS: Real-time Target Speaker Extraction with Lightweight Conv-TasNet and State Space Modeling', 'sato24_interspeech', 'tse ssm frontend depen
11146dency convolutional encoder tasnet complexity computational enlarge'], ['Jessica Monaghan, Arun Sebastian, Nicky Chong-White, Vicky Zhang, Vijayalakshmi Easwar, Padraig Kitterick', "Automatic Detection of Hearing Loss from Children's Speech using wav2vec 2.0 Features", 'monaghan24_interspeech', 'early scalable developmental xgboost acknowledging preschool proof-of-concept non-intrusive screening toward'], ['Iuliia Zaitova, Irina Stenger, Wei Xue, Tania Avgustinova, Bernd Möbius, Dietrich Klakow', 'Cross-Linguistic Intelligibility of Non-Compositional Expressions in Spoken Context', 'zaitova24_interspeech', 'russian surprisal ukrainian distance phonological multiple-choice slavic bulgarian polish score'], ['Arnon Turetzky, Or Tal, Yael Segal, Yehoshua Dissen, Ella Zeldes, Amit Roth, Eyal Cohen, Yosi Shrem, Bronya R. Chernyak, Olga Seleznova, Joseph Keshet, Yossi Adi', 'HebDB: a Weakly Supervised Dataset for Hebrew Speech Processing', 'turetzky24_interspeech', 'language pre-processed model recording asr provide multi-lingual spoken page baseline'], ['Wei Xue, Ivan Yuen, Bernd Möbius', 'Towards a better understanding of receptive multilingualism: listening conditions and priming effects', 'xue24_interspeech', 'awr similarity affected comprehend word null less prime language-dependent without'], ['Zhiyong Yan, Heinrich Dinkel, Yongqing Wang, Jizhong Liu, Junbo Zhang, Yujun Wang, Bin Wang', 'Bridging Language Gaps in Audio-Text Retrieval', 'yan24_interspeech', 'text encoder ced clotho audiocaps excels content abundance disparity non-english'], ['Xin Wang, Tomi Kinnunen, Kong Aik Lee, Paul-Gauthier Noé, Junichi Yamagishi', 'Revisiting and Improving Scoring Fusion for Spoofing-aware Speaker Verification Using Compositional Data Analysis', 'wang24l_interspeech', 'asv score-level calibration spoofing non-linear finding score zero-effort decision summing'], ['Miseul Kim, Soo-Whan Chung, Youna Ji, Hong-Goo Kang, Min-Seok Choi', 'Speak in the Scene: Diffusion-based Acoustic Scene Transfer toward Immersive Speech Generation', 'kim24e_interspeech', 'ast environment target clap signal task diffusion audio emphasize accompanied'], ['Haiyang Sun, Fulin Zhang, Yingying Gao, Shilei Zhang, Zheng Lian, Junlan Feng', 'MFSN: Multi-perspective Fusion Search Network For Pre-training Knowledge in Speech Emotion Recognition', 'sun24b_interspeech', 'comprehensiveness emotional appropriateness sec ser capturing overlooking content cue verifies'], ['Hyun Kyung Hwang, Manami Hirayama', 'Acquisition of high vowel devoicing in Japanese: A production experiment with three and four year olds', 'hwang24b_interspeech', 'hvd developmental word-medial position advancement old rate pattern distinct age'], ['Rotem Rousso, Eyal Cohen, Joseph Keshet, Eleanor Chodroff', 'Tradition or Innovation: A Comparison of Modern ASR Methods for Forced Alignment', 'rousso24_interspeech', 'whisperx mm mfa gmm-hmm speech dominantly kaldi-based montreal massively buckeye'], ['Young Jin Ahn, Jungwoo Park, Sangha Park, Jonghyun Choi, Kee-Eung Kim', 'SyncVSR: Data-Efficient Visual Speech Recognition with End-to-End Crossmodal Audio Token Synchronization', 'ahn24_interspeech', 'vsr synchronizes fell versatility intersection visemes non-autoregressive sought stand aligning'], ['Fengrun Zhang, Wangjin Zhou, Yiming Liu, Wang Geng, Yahui Shan, Chen Zhang', 'Disentangling Age and Identity with a Mutual Information Minimization for Cross-Age Speaker Verification', 'zhang24c_interspeech', 'embeddings backbone gap vox-ca identity-related disentangled age-related aging disentangle method'], ['Naoki Makishima, Naotaka Kawata, Mana Ihori, Tomohiro Tanaka, Shota Orihashi, Atsushi Ando, Ryo Masumura', 'SOMSRED: Sequential Output Modeling for Joint Multi-talker Overlapped Speech Recognition and Speaker Diarization', 'makishima24_interspeech', 'identifier asr unknown jointly fully separate timestamps clustering-based recursively sub-optimal'], ['Xihang Qiu, Lixian Zhu, Zikai Song, Zeyu Chen, Haojie Zhang, Kun Qian, Ye Zhang, Bin Hu, Yoshiharu Yamamoto, Björn W. Schuller', 'Study Selectively: An Adaptive Knowledge Distillation based on a Voting Network for Heart Sound Classification', 'qiu24_interspeech', 'excellent teacher student impart complexity sizeable computational strategy tell nowadays'], ['Takafumi Moriya, Takanori Ashihara, Masato Mimura, Hiroshi Sato, Kohei Matsuura, Ryo Masumura, Taichi Asami', 'Boosting Hybrid Autoregressive Transducer-based ASR with Internal Acoustic Model Training and Dual Blank Thresholding', 'moriya24_interspeech', 'hat iam non-blank decoding transducer joint emit synchronously speed-up skip'], ['I-Ting Hsieh, 
11146Chung-Hsien Wu', 'Dysarthric Speech Recognition Using Curriculum Learning and Articulatory Feature Embedding', 'hsieh24_interspeech', 'characteristic recognizing disorder diverse incorporate patient speaker commonly efficiency additionally'], ['Matthijs Van keirsbilck, Alexander Keller', 'Conformer without Convolutions', 'vankeirsbilck24_interspeech', 'learnable temporal surprising speech-to-text averaging discover replace completely remove shift'], ['Yi-Cheng Lin, Tzu-Quan Lin, Hsi-Che Lin, Andy T. Liu, Hung-yi Lee', 'On the social bias of speech self-supervised models', 'lin24b_interspeech', 'ssl biased debiasing inadvertently marginalized reinforcing amplify disparate training shallower'], ['Ke-Han Lu, Zhehuai Chen, Szu-Wei Fu, He Huang, Boris Ginsburg, Yu-Chiang Frank Wang, Hung-yi Lee', 'DeSTA: Enhancing Speech Language Models through Descriptive Speech-Text Alignment', 'lu24c_interspeech', 'slms instruction-following capability reshape generalizing captioning non-linguistic caption zero-shot facilitating'], ['Ching-Yu Yang, Shreya G. Upadhyay, Ya-Tse Wu, Bo-Hao Su, Chi-Chun Lee', 'RW-VoiceShield: Raw Waveform-based Adversarial Attack on One-shot Voice Conversion', 'yang24e_interspeech', 'protected speaker utterance attacking imperceptible white-box recognizable generated undergoes disparity'], ['Xinwei Cao, Zijian Fan, Torbjørn Svendsen, Giampiero Salvi', 'A Framework for Phoneme-Level Pronunciation Assessment Using CTC', 'cao24b_interspeech', 'gop alignment method prone traditional deletion insertion error child alignment-based'], ['Woan-Shiuan Chien, Chi-Chun Lee', 'An Investigation of Group versus Individual Fairness in Perceptually Fair Speech Emotion Recognition', 'chien24_interspeech', 'ser raters gender label standpoint diminished persist issue biased arising'], ['Martin Lenglet, Olivier Perrotin, Gerard Bailly', 'FastLips: an End-to-End Audiovisual Text-to-Speech System with Lip Features Prediction for Virtual Avatars', 'lenglet24_interspeech', 'fastspeech avatar facial explicit co-verbal encoder model movement generation visual'], ['Yin-Tse Lin, Shreya G. Upadhyay, Bo-Hao Su, Chi-Chun Lee', 'SWiBE: A Parameterized Stochastic Diffusion Process for Noise-Robust Bandwidth Expansion', 'lin24c_interspeech', 'score-based bwe parameterizations stepwise gan-based hyperparameters current including manifest encounter'], ['Joun Yeop Lee, Myeonghun Jeong, Minchan Kim, Ji-Hyun Lee, Hoon-Young Cho, Nam Soo Kim', 'High Fidelity Text-to-Speech Via Discrete Tokens Using Token Transducer and Group Masked Language Model', 'lee24f_interspeech', 'interpreting semantic speech speaking stage module enriching conformer-based high-fidelity text'], ['Shreya G. Upadhyay, Carlos Busso, Chi-Chun Lee', 'A Layer-Anchoring Strategy for Enhancing Cross-Lingual Speech Emotion Recognition', 'upadhyay24_interspeech', 'ser layer pretrained transformer hierarchical uncovers encapsulates msp-podcast podcast model'], ['Wei-Tung Hsu, Chin-Po Chen, Yun-Shao Lin, Chi-Chun Lee', 'A Cluster-based Personalized Federated Learning Strategy for End-to-End ASR of Dementia Patients', 'hsu24_interspeech', 'heterogeneity usage pause asr-related wer privacy-preserving adress distribution alzheimer challenge'], ['Hsing-Hang Chou, Woan-Shiuan Chien, Ya-Tse Wu, Chi-Chun Lee', 'An Inter-Speaker Fairness-Aware Speech Emotion Regression Framework', 'chou24_interspeech', 'ser fairness id fair knowing speaker cluster upfront speaker-level individual'], ['Jan Lehečka, Josef V. Psutka, Lubos Smidl, Pavel Ircing, Josef Psutka', 'A Comparative Analysis of Bilingual and Trilingual Wav2Vec Models for Automatic Speech Recognition in Multilingual Oral History Archives', 'lehecka24_interspeech', 'mixed-language archive monolingual unique public commonvoice heritage releasing dataset push'], ['Dirk Eike Hoffner, Jana Roßbach, Bernd T. Meyer', 'Joint prediction of subjective listening effort and speech intelligibility based on end-to-end learning', 'hoffner24_interspeech', 'non-intrusive character root-mean-square entropy-based model intrusive hearing-impaired normal-hearing quantified real-life'], ['Yaroslav Getman, Tamas Grosz, Mikko Kurimo', 'What happens in continued pre-training? Analysis of self
11146-supervised speech models with continued pre-training for colloquial Finnish ASR', 'getman24_interspeech', 'multilingual specialize less-resourced high-resourced codeword language discovered quantized foundation purely'], ['Fredrik Cumlin, Xinyu Liang, Victor Ungureanu, Chandan K. A. Reddy, Christian Schüldt, Saikat Chatterjee', 'DNSMOS Pro: A Reduced-Size DNN for Probabilistic MOS of Speech', 'cumlin24_interspeech', 'non-intrusive subjectively datasets rated quality method training design voip architecture'], ['Yaroslav Getman, Tamas Grosz, Katri Hiovain-Asikainen, Mikko Kurimo', 'Exploring adaptation techniques of large speech foundation models for low-resource ASR: a case study on Northern Sámi', 'getman24b_interspeech', 'fine-tuning pre-training already extended include augments low-resourced new preparing continued'], ['Hoan My Tran, David Guennec, Philippe Martin, Aghilas Sini, Damien Lolive, Arnaud Delhay, Pierre-François Marteau', 'Spoofed Speech Detection with a Focus on Speaker Embedding', 'tran24_interspeech', 'deepfakes contrastive excel loss layer-wise deepfake finetuned spoof attentive wavlm'], ['Ya-Tse Wu, Jingyao Wu, Vidhyasaharan Sethu, Chi-Chun Lee', 'Can Modelling Inter-Rater Ambiguity Lead To Noise-Robust Continuous Emotion Predictions?', 'wu24d_interspeech', 'noise cer insufficiently ccc recola robustness concordance regularize broadly arousal'], ['Yuanyuan Zhang, Zhengjun Yue, Tanvina Patel, Odette Scharenborg', 'Improving child speech recognition with augmented child-like speech', 'zhang24d_interspeech', 'child-to-child cross-lingual augmentation model absolute asrs two-fold asr csr suboptimal'], ['Victor Miara, Theo Lepage, Reda Dehak', 'Towards Supervised Performance on Speaker Verification with Self-Supervised Learning by Leveraging Large-Scale ASR Models', 'miara24_interspeech', 'ssl pseudo-labels fine-tuning eer narrowing wavlm representation establishing iteratively refined'], ['Guanrou Yang, Ziyang Ma, Fan Yu, Zhifu Gao, Shiliang Zhang, Xie Chen', 'MaLa-ASR: Multimedia-Assisted LLM-Based ASR', 'yang24f_interspeech', 'llm auxiliary audio integrate sparked keywords ingest information-rich surge fresh'], ['Cécile Macaire, Chloé Dion, Didier Schwab, Benjamin Lecouteux, Emmanuelle Esperança-Rodier', 'Towards Speech-to-Pictograms Translation', 'macaire24_interspeech', 'cascade end-to-end tailor speech state-of-the-art in-depth nlp specially everyday released'], ['June-Woo Kim, Miika Toikkanen, Yera Choi, Seoung-Eun Moon, Ho-Young Jung', 'BTS: Bridging Text and Sound Modalities for Metadata-Aided Respiratory Sound Classification', 'kim24f_interspeech', 'rsc metadata text-audio patient recording multimodal icbhi sample surpassing validates'], ['Alexander Kathan, Martin Bürger, Andreas Triantafyllopoulos, Sabrina Milkus, Jonas Hohmann, Pauline Muderlak, Jürgen Schottdorf, Richard Musil, Björn Schuller, Shahin Amiriparian', 'Real-world PTSD Recognition: A Cross-corpus and Cross-linguistic Evaluation', 'kathan24_interspeech', 'mental analyse early disaster post-traumatic abuse wellbeing sexual cross-cultural combat'], ['Pu Wang, Junhui Li, Jialu Li, Liangdong Guo, Youshan Zhang', 'Diffusion Gaussian Mixture Audio Denoise', 'wang24m_interspeech', 'model reverse noise distribution signal clean comply subtracted noisy estimate'], ['Thomas Graave, Zhengyang Li, Timo Lohrenz, Tim Fingscheidt', 'Mixed Children/Adult/Childrenized Fine-Tuning for Children’s ASR: How to Reduce Age Mismatch and Speaking Style Mismatch', 'graave24_interspeech', 'speech read partially matched pre-trained child absolute datasets individual deteriorate'], ['Shucong Zhang, Titouan Parcollet, Rogier van Dalen, Sourav Bhattacharya', 'Linear-Complexity Self-Supervised Learning for Speech Processing', 'zhang24e_interspeech', 'mhsa pre-training ssl gpus week wav vec encoder model multi-headed'], ['Fan Huang, Kun Zeng, Wei Zhu', 'DiffVC+: Improving Diffusion-based Voice Conversion for Speaker Anonymization', 'huang24_interspeech', 'privacy converted speech encoder embedding server-side content suppresses decoupled leakage'], ['Zhengyang Li, Patrick Blumenberg, Jing Liu, Thomas Graave, Timo Lohrenz, Siegfried Kunzmann, Tim Fingscheidt', 'Interleaved Audio/Audiovisual Transfer Learning for AV-ASR in Low-Resourced Languages', 'li24i_interspeech', '-stage target stage language german excels cross-modality catastrophic forgetting english'], ['Arnav Kundu, Prateeth Nayak, Priyanka Padmanabhan, Devang Naik', 'RepCNN: Micro-sized, Mighty Models for Wakeword Detection', 'kundu24_interspeech', 'always-on runtime memory footprint compute inference convolutional model re-parameterized parameter'], ['Théodor Lemerle, Nicolas Obin, Axel Roebel', 'Small-E: Small Language Model with Linear Attention for Efficient Speech Synthesis', 'lemerle24_interspeech', 'transformer cloning zero-shot architecture showcased decoder-only powered impeding tt skipping'], ['Ji-Hun Kang, Jae-Hong Lee, Mun-Hak Lee, Joon-Hyuk Chang', 'Whisper Multilingual Downstream Task Tuning Using Task Vectors', 'kang24_interspeech', 'model vector direction weight orient summing simple space arithmetic effective'], ['Luis Felipe Parra-Gallego, Tilak Purohit, Bogdan Vlasenko, Juan Rafael Orozco-Arroyave, Mathew Magimai.-Doss', 'Cross-transfer Knowledge between Speech and Text Encoders to Evaluate Customer Satisfaction', 'parragallego24_interspeech', 'bert reputation distilling enriching cost-effective asr wavlm learning company whisper'], ['Honglie Chen, Rodrigo Mira, Stavros Petridis, Maja Pantic', 'RT-LA-VocE: Real-Time Low-SNR Audio-Visual Speech Enhancement', 'chen24g_interspeech', 'frame causal latency stream emformer devising non-causal state-of-the-art audio frame-by-frame'], ['Mara Barberis, Pieter De Clercq, Bastiaan Tamm, Hugo Van hamme, Maaike Vandermosten', 'Automatic recognition and detection of aphasic natural speech', 'barberis24_interspeech', 'aphasia asr svm feature detect stroke semi-automatically administered semi-automatic consuming'], ['Moreno La Quatra, Maria Francesca Turco, Torbjørn Svendsen, Giampiero Salvi, Juan Rafael Orozco-Arroyave, Sabato Marco Siniscalchi', "Exploiting Foundation Models and Speech Enhancement for Parkinson's Disease Detection from Speech in Real-World Operative Conditions", 'laquatra24_interspeech', 'pc-gita base performance devising foundational assess off-the-shelf wavlm hubert fine-tune'], ['Cong Zhang, Tong Li, Gayle DeDe, Christos Salis', 'Prosody of speech production in latent post-stroke aphasia', 'zhang24f_interspeech', 'neurotypical mild forest prosodic random left-hemisphere utterance-initial reinfor
11146ced stroke control'], ['Lorenzo Maselli, Véronique Delvaux', 'Aerodynamics of Sakata labial-velar oral stops', 'maselli24_interspeech', 'bilabial airflow plain pressure congo delineated variable université southwestern mon'], ['Jongsuk Kim, Jiwon Shin, Junmo Kim', 'AVCap: Leveraging Audio-Visual Features as Text Tokens for Captioning', 'kim24g_interspeech', 'advancement extensibility human-level pivotal scalability exploration com github model applicable'], ['Hassan Taherian, Vahid Ahmadi Kalkhorani, Ashutosh Pandey, Daniel Wong, Buye Xu, DeLiang Wang', 'Towards Explainable Monaural Speaker Separation with Auditory-based Training', 'taherian24_interspeech', 'pit permutation onset criterion cue ambiguity pitch hybrid talker-independent same-gender'], ['Anith Selvakumar, Homa Fashandi', 'Getting More for Less: Using Weak Labels and AV-Mixup for Robust Audio-Visual Speaker Verification', 'selvakumar24_interspeech', 'dml voxceleb multimodal overfit space multitask reporting owing dominated learning'], ['Avihu Dekel, Raul Fernandez', 'Exploring the Benefits of Tokenization of Discrete Acoustic Units', 'dekel24_interspeech', 'variable-rate vocabulary audio-based overlooked showcase task merge playing grapheme-to-phoneme explanation'], ['Junzuo Zhou, Jiangyan Yi, Tao Wang, Jianhua Tao, Ye Bai, Chu Yuan Zhang, Yong Ren, Zhengqi Wen', 'TraceableSpeech: Towards Proactively Traceable Text-to-Speech with Watermarking', 'zhou24b_interspeech', 'watermark imperceptibility speech attack flexibility quality watermarked vall-e tt resilience'], ['Georgios Chochlakis, Chandrashekhar Lavania, Prashant Mathur, Kyu J. Han', 'Tackling Missing Modalities in Audio-Visual Representation Learning Using Masked Autoencoders', 'chochlakis24_interspeech', 'imputed pristine robust curriculum make lombard retraining trained autoencoder necessarily'], ['Lun Wang, Om Thakkar, Zhong Meng, Nicole Rafidi, Rohit Prabhavalkar, Arun Narayanan', 'Efficiently Train ASR Models that Memorize Less and Perform Better with Per-core Clipping', 'wang24n_interspeech', 'pcc gradient memorization unintended mitigate apcc minibatch streamlined multifaceted training'], ['Zilong Huang, Man-Wai Mak, Kong Aik Lee', 'MM-NodeFormer: Node Transformer Multimodal Fusion for Emotion Recognition in Conversation', 'huang24b_interspeech', 'modality erc richness emotional auxiliary text consultation visual meld main'], ['Ji Sub Um, Hoirin Kim', 'Utilizing Adaptive Global Response Normalization and Cluster-Based Pseudo Labels for Zero-Shot Voice Conversion', 'um24b_interspeech', 'content information conduct speaker convnext layer dynamic transmit conveying transmitted'], ['Jialu Li, Mark Hasegawa-Johnson, Karrie Karahalios', 'Enhancing Child Vocalization Classification with  Phonetically-Tuned Embeddings for Assisting Autism Diagnosis', 'li24j_interspeech', 'clinician old auxiliary behavior build year reproducible labor helping audio'], ['Alexander Johnson, Peter Plantinga, Pheobe Sun, Swaroop Gadiyaram, Abenezer Girma, Ahmad Emami', 'Efficient SQA from Long Audio Contexts: A Policy-driven Approach', 'johnson24_interspeech', 'file compute safely skipping infeasible podcasts recording skip retrieving less'], ['Li-Fang Lai, Nicole Holliday', 'Voice Quality Variation in AAE: An Additional Challenge for Addressing Bias in ASR Models?', 'lai24_interspeech', 'creaky mae creak american men woman young phonation creakier non-modal'], ['Shiran Aziz, Yossi Adi, Shmuel Peleg', 'Audio Enhancement from Multiple Crowdsourced Recordings: A Simple and Effective Baseline', 'aziz24_interspeech', 'event local noise location device signal cleaned uncorrelated cellular unrelated'], ['Wen Wu, Chao Zhang, Philip C. Woodland', 'Confidence Estimation for Automatic Detection of Depression and Alzheimer’s Disease Based on Clinical Interviews', 'wu24e_interspeech', 'daic-woz informs adress distribution second-order dirichlet clinician attracted speech-based risk'], ['Vikentii Pankov, Valeria Pronina, Alexander Kuzmin, Maksim Borisov, Nikita Usoltsev, Xingshan Zeng, Alexander Golubkov, Nikolai Ermolenko, Aleksandra Shirshova, Yulia Matveeva', 'DINO-VITS: Data-Efficient Zero-Shot TTS with Self-Supervised Speaker Verification Loss for Noise Robustness', 'pankov24_interspeech', 'dino noisy training encoder clean noise-robustness speech objective cloning discriminability'], ['Jie Chi, Electra Wallington, Peter Bell', 'Characterizing code-switching: Applying Linguistic Principles for Metric Assessment and Development', 'chi24_interspeech', 'leverage data-sets hindi-english capture mandarin-english code-switched assess ric
11146hness intuition participating'], ['Sean Robertson, Gerald Penn, Ewan Dunbar', 'Quantifying the Role of Textual Predictability in Automatic Speech Recognition', 'robertson24_interspeech', 'model asr straightforwardly ability long-standing diagnosing use higher-order spite context'], ['Shiyi Han, Mingbin Xu, Zhihong Lei, Zhen Huang, Xingyu Na', 'Enhancing CTC-based speech recognition with diverse modeling units', 'han24_interspeech', 'model synergistic grapheme-based asr phoneme-based heterogeneous driving evolution align accuracy'], ['Zhiqi Huang, Diamantino Caseiro, Kandarp Joshi, Christopher Li, Pat Rondon, Zelin Wu, Petr Zadrazil, Lillian Zhou', 'Optimizing Large-Scale Context Retrieval for End-to-End ASR', 'huang24c_interspeech', 'entity scalable comparative recall scoring contextual accurate outstanding segment-level method'], ['Yiling Huang, Weiran Wang, Guanlong Zhao, Hank Liao, Wei Xia, Quan Wang', 'On the Success and Limitations of Auxiliary Network Based Word-Level End-to-End Neural Speaker Diarization', 'huang24d_interspeech', 'quot spoke asr turn-based capability modularized logits eend blank -speaker'], ['Chung-Ming Chien, Andros Tjandra, Apoorv Vyas, Matt Le, Bowen Shi, Wei-Ning Hsu', 'Learning Fine-Grained Controllability on Speech Generation via Efficient Fine-Tuning', 'chien24b_interspeech', 'voicebox adapter pre-trained module lora grow reuse across compromising follow-up'], ['Ankit Gupta, George Saon, Brian Kingsbury', 'Exploring the limits of decoder-only models trained on public speech recognition corpora', 'gupta24_interspeech', 'whisper asr transformer competitive open hour owsm permissive large-v checkpoint'], ['Woo Hyun Kang, Srikanth Vishnubhotla, Rudolf Braun, Yogesh Virkar, Raghuveer Peri, Kyu J. Han', 'SWAN: SubWord Alignment Network for HMM-free word timing estimation in end-to-end automatic speech recognition', 'kang24b_interspeech', 'wte hmm asr multilingual method label fleurs inability accumulated due'], ['Frank Seide, Yangyang Shi, Morrie Doulaty, Yashesh Gaur, Junteng Jia, Chunyang Wu', 'Speech ReaLLM – Real-time Speech Recognition with Multimodal Language Models by Teaching the Flow of Time', 'seide24_interspeech', 'llm decoder-only architecture rnn-t asr end-pointing empty real reasonably without'], ['Tuan Vu Ho, Kota Dohi, Yohei Kawaguchi', 'Stream-based Active Learning for Anomalous Sound Detection in Machine Condition Monitoring', 'ho24_interspeech', 'asd sample budget unexplored dcase updating framework receiver backend retraining'], ['Virat Shejwalkar, Om Thakkar, Arun Narayanan', 'Quantifying Unintended Memorization in BEST-RQ ASR Encoders', 'shejwalkar24_interspeech', 'auditing downstream real-world unintentionally memorize sample clipping impacting conformer-based mitigating'], ['Ke Chen, Jiaqi Su, Taylor Berg-Kirkpatrick, Shlomo Dubnov, Zeyu Jin', 'Improving Generalization of Speech Separation in Real-World Scenarios: Strategies in Simulation, Optimization, and Evaluation', 'chen24h_interspeech', 'pipeline training diverse content acoustic separator pit objective across environment'], ['Tien-Ju Yang, Andrew Rosenberg, Bhuvana Ramabhadran', 'Contemplative Mechanism for Speech Recognition: Speech Encoders can Think', 'yang24g_interspeech', 'quot token size encoder strategically interleaving doubling model inserting encourages'], ['Pan-Pan Jiang, Jimmy Tobin, Katrin Tomanek, Robert MacDonald, Katie Seaver, Richard Cave, Marilyn Ladewig, Rus Heywood, Jordan Green', 'Learnings from curating a trustworthy, well-annotated, and useful dataset of disordered English speech', 'jiang24_interspeech', 'metadata project correction transcript annotation inter-rater report machine-learning gathering rationale'], ['Ante Jukić, Roman Korostik, Jagadeesh Balam, Boris Ginsburg', 'Schrödinger Bridge for Generative Speech Enhancement', 'jukic24_interspeech', 'model dereverberation denoising clean loss complex-valued diffusion-based distribution tractable baseline'], ['Naijun Zheng, Xucheng Wan, Kai Liu, Ziqing Du, Zhou Huan', 'An efficient text augmentation approach for contextualized Mandarin speech recognition', 'zheng24_interspeech', 'contextualize speech-text text-only asr codebook embeddings pre-trained uncommon top-performing hindered'], ['Jiafeng Zhong, Bin Li, Jiangyan Yi', 'Enhancing Parti
11146ally Spoofed Audio Localization with Boundary-aware Attention Mechanism', 'zhong24_interspeech', 'boundary bam authenticity frame partialspoof intra-frame inter-frame fake unexplored frame-wise'], ['Zejiang Hou, Goeric Huybrechts, Anshu Bhatia, Daniel Garcia-Romero, Kyu J. Han, Katrin Kirchhoff', 'Revisiting Convolution-free Transformer for Speech Recognition', 'hou24_interspeech', 'conformer convolution wer architecture training scale state-ofthe-art module catch inductive'], ['Houjian Guo, Chaoran Liu, Carlos Toshinori Ishi, Hiroshi Ishiguro', 'X-E-Speech: Joint Training Framework of Non-Autoregressive Cross-lingual Emotional Text-to-Speech and Voice Conversion', 'guo24b_interspeech', 'tt speech model style-related content-related freeze similarity content nar synthesis'], ['Haici Yang, Jiaqi Su, Minje Kim, Zeyu Jin', 'Genhancer: High-Fidelity Speech Enhancement via Generative Modeling on Discrete Codec Tokens', 'yang24h_interspeech', 'conditioning best-fit speaker-identity content retention enforce conventional audio domain supplement'], ['Yanxiong Li, Jiaxin Tan, Guoqing Chen, Jialong Li, Yongjie Si, Qianhua He', 'Low-Complexity Acoustic Scene Classification Using Parallel Attention-Convolution Network', 'li24k_interspeech', 'dcase challenge contextual global local exceeds distillation method clip pre-processing'], ['Yifei Xin, Zhihong Zhu, Xuxin Cheng, Xusheng Yang, Yuexian Zou', 'Audio-text Retrieval with Transformer-based Hierarchical Alignment and Disentangled Cross-modal Representation', 'xin24b_interspeech', 'atr dcr tha latent factor transformer audio finegrained semantic single-level'], ['Huai-Zhe Yang, Chia-Ping Chen, Shan-Yun He, Cheng-Ruei Li', 'Bilingual and Code-switching TTS Enhanced with Denoising Diffusion Model and GAN', 'yang24i_interspeech', 'adversarial consistency mo speaker employ classifier speech mandarin-english push featuring'], ['Shiu-Hsiang Liou, Po-Cheng Chan, Chia-Ping Chen, Tzu-Chieh Lin, Chung-Li Lu, Yu-Han Cheng, Hsiang-Feng Chuang, Wei-Yu Chen', 'Enhancing ECAPA-TDNN with Feature Processing Module and Attention Mechanism for Speaker Verification', 'liou24_interspeech', 'kernel temporal front-end weight ecapa capture flop assigns expands mindcf'], ['HengYu Li, Kangdi Mei, Zhaoci Liu, Yang Ai, Liping Chen, Jie Zhang, Zhenhua Ling', 'Refining Self-supervised Learnt Speech Representation using Brain Activations', 'li24l_interspeech', 'similarity model downstream pre-trained often-used superb fmri human aligning refine'], ['Tianteng Gu, Bei Liu, Hang Shao, Yanmin Qian', 'SparseWAV: Fast and Accurate One-Shot Unstructured Pruning for Large Speech Foundation Models', 'gu24_interspeech', 'parameter remove compression pre-trained unimportant eliminated sacrificing negligible performance consumption'], ['Changhwan Kim', 'ClariTTS: Feature-ratio Normalization and Duration Stabilization for Code-mixed Multi-speaker Speech Synthesis', 'kim24h_interspeech', 'tt accent language speaker synthesized disentangles flow-based speaker-related linguistic affine'], ['Junhui Li, Pu Wang, Jialu Li, Youshan Zhang', 'Complex Image-Generative Diffusion Transformer for Audio Denoising', 'li24m_interspeech', 'attention image field generation deep model expands leaving receptive scalability'], ['Aijun Li, Jun Gao, Zhiwei Wang', 'Effect of Complex Boundary Tones on Tone Identification: An Experimental Study with Mandarin-speaking Preschool Children', 'li24n_interspeech', 'suabt preschooler word age tonal forty-eight decoding stabilizes impede child-directed'], ['Nameer Hirschkind, Xiao Yu, Mahesh Kumar Nandwana, Joseph Liu, Eloi DuBois, Dao Le, Nicolas Thiebaut, Colin Sinclair, Kyle Spence, Charles Shang, Zoe Abrams, Morgan McGuire', 'Diffusion Synthesizer for Efficient Multilingual Speech to Speech Translation', 'hirschkind24_interspeech', 'diffusion-based tacotron-based low-latency translating zero-shot double speech-to-speech bleu count pesq'], ['Yu Nakagome, Michael Hentschel', 'InterBiasing: Boost Unseen Word Recognition through Biasing Intermediate Predictions', 'nakagome24_interspeech', 'label ctc subsequent unknown self-conditioned parameter-free keywords layer pair substituting'], ['Haixin Guan, Wei Dai, Guangyong Wang, Xiaobin Tan, Peng Li, Jiaen Liang', 'Reducing Speech Distortion and Artifacts for Speech Enhancement by Loss Function', 'guan24_interspeech', 'combined governing diminish stride persist dns lightweight learning-based continuity formation'], ['Tzu-Quan 
11146Lin, Hung-yi Lee, Hao Tang', 'DAISY: Data Adaptive Self-Supervised Early Exit for Speech Representation Models', 'lin24d_interspeech', 'fine-tuning inference adaptivity layer need round decides hubert eliminating adjusting'], ['Guanlin Chen, Yun Jin', "Cascaded Transfer Learning Strategy for Cross-Domain Alzheimer's  Disease Recognition through Spontaneous Speech", 'chen24i_interspeech', 'subspace second-level first-level corpus gpt- multi-source respectively experiment forest align'], ['Keiko Ochi, Koji Inoue, Divesh Lala, Tatsuya Kawahara', 'Entrainment Analysis and Prosody Prediction of Subsequent Interlocutor’s Backchannels in Dialogue', 'ochi24_interspeech', 'preceding utterance prosodic feature power regression interrelationship empathy svr attentive'], ['Shubham Gupta, Mirco Ravanelli, Pascal Germain, Cem Subakan', 'Phoneme Discretized Saliency Maps for Explainable Detection of AI-Generated Voice', 'gupta24b_interspeech', 'explanation discretization faithful understandable associating fastspeech standard tacotron algorithm experimentally'], ['Atsushi Ando, Takafumi Moriya, Shota Horiguchi, Ryo Masumura', 'Factor-Conditioned Speaking-Style Captioning', 'ando24_interspeech', 'caption generates gts factor ensure diverse deterministically original learning guarantee'], ['Zhe Liu, Suyoun Kim, Ozlem Kalinli', 'Evaluating Speech Recognition Performance Towards Large Language Model Based Voice Assistants', 'liu24c_interspeech', 'llm asr semantic metric curated popularity failure cascaded raised judgement'], ['Francesco Paissan, Luca Della Libera, Zhepei Wang, Paris Smaragdis, Mirco Ravanelli, Cem Subakan', 'Audio Editing with Non-Rigid Text Prompts', 'paissan24b_interspeech', 'edits diffusion pipeline latent inpainting faithfulness lora user qualitatively fidelity'], ['Shilin Wang, Haixin Guan, Yanhua Long', 'QMixCAT: Unsupervised Speech Enhancement Using Quality-guided Signal Mixing and Competitive Alternating Model Training', 'wang24o_interspeech', 'teacher-student supervised mixture cat epoch learning-based framework iteratively trained innovative'], ['Shuai Wang, Dehao Zhang, Kexin Shi, Yuchen Wang, Wenjie Wei, Jibin Wu, Malu Zhang', 'Global-Local Convolution with Spiking Neural Networks for Energy-efficient Keyword Spotting', 'wang24p_interspeech', 'kw module fewer efficiency energy extraction sparser hand-crafted lightweight performance'], ['Yuexuan Kong, Viet-Anh Tran, Romain Hennequin', 'STraDa: A Singer Traits Dataset', 'kong24_interspeech', 'metadata downloadable track lead thousand file bias ssc twenty-five benchmarked'], ['Junseok Ahn, Youkyum Kim, Yeunju Choi, Doyeop Kwak, Ji-Hoon Kim, Seongkyu Mun, Joon Son Chung', 'VoxSim: A perceptual voice similarity dataset', 'ahn24b_interspeech', 'speaker prediction vcc utilised benchmarking automate unexplored score leaving out-of-domain'], ['Jialong Mai, Xiaofen Xing, Weidong Chen, Xiangmin Xu', 'DropFormer: A Dynamic Noise-Dropping Transformer for Speech Emotion Recognition', 'mai24_interspeech', 'dropping emotional local information ser token non-emotional overlook meld overly'], ['Peng Wang, Yifan Yang, Zheng Liang, Tian Tan, Shiliang Zhang, Xie Chen', 'Incorporating Class-based Language Model for Named Entity Recognition in Factorized Neural Transducer', 'wang24q_interspeech', 'fnt ner biasing hurting decoupling triggering excessive lm attention-based risk'], ['Duc-Tuan Truong, Ruijie Tao, Tuan Nguyen, Hieu-Thi Luong, Kong Aik Lee, Eng Siong Chng', 'Temporal-Channel Modeling in Multi-head Self-Attention for Synthetic Speech Detection', 'truong24b_interspeech', 'mhsa temporal transformer dependency channel module neglect ablation input artifact'], ['Fei Zhao, Chenggang Zhang, Shulin He, Jinjiang Liu, Xueliang Zhang', 'Deep Echo Path Modeling for Acoustic Echo Cancellation', 'zhao24_interspeech', 'aec near-end scenario learning unseen single-talk complex full-duplex double-talk generalizing'], ['Jen-Hung Huang, Wei-Tsung Lee, Chung-Hsien Wu', 'USD-AC: Unsupervised Speech Disentanglement for Accent Conversion', 'huang24e_interspeech', 'content generalization disentangles exceptional linguistic adaptability grounded generalizability boosting learning'], ['Yuqin Lin, Longbiao Wang, Jianwu Dang, Nobuaki Minematsu', 'Exploring Pre-trained Speech Model for Articulatory Feature Extraction in Dysarthric Speech Using ASR', 'lin24e_interspeech', 'ph
11146onemic dysarthria pretrained attribute extract torgo detection uaspeech information dysphonia'], ['Yiying Hu, Hui Feng', 'Key Acoustic Cues for the Realization of Metrical Prominence in Tone Languages: A Cross-Dialect Study', 'hu24_interspeech', 'right-dominant pitch cumulative analysis dialect intensity identify metrically dynamic across'], ['Marc Härkönen, Samuel J. Broughton, Lahiru Samarakoon', 'EEND-M2F: Masked-attention mask transformers for speaker diarization', 'harkonen24_interspeech', 'end-to-end dihard-iii segmentation alimeeting winning truly stack eliminating irrelevant der'], ['Dongheon Lee, Jung-Woo Choi', 'DeFTAN-AA: Array Geometry Agnostic Multichannel Speech Enhancement', 'lee24g_interspeech', 'spatial cross-attention microphone block channel-wise foreground alleviates dense gated various'], ['Neil Shah, Shirish Karande, Vineet Gandhi', 'Towards Improving NAM-to-Speech Synthesis Intelligibility using Self-Supervised Speech Models', 'shah24_interspeech', 'nam self-supervision seq ground-truth methodology non-audible mcd murmur cstr gauge'], ['Pin-Yen Liu, Jen-Tzung Chien', 'Modality Translation Learning for Joint Speech-Text Model', 'liu24d_interspeech', 'speech shared text emotional cross-modality harness space unexplored struggle hardly'], ['Semin Kim, Myeonghun Jeong, Hyeonseung Lee, Minchan Kim, Byoung Jin Choi, Nam Soo Kim', 'MakeSinger: A Semi-Supervised Training Method for Data-Efficient Singing Voice Synthesis via Classifier-free Diffusion Guidance', 'kim24i_interspeech', 'svs data pitch tt semisupervised diffusion-based gathering guiding text dual'], ['Hanyu Meng, Qiquan Zhang, Xiangyu Zhang, Vidhyasaharan Sethu, Eliathamby Ambikairajah', 'Binaural Selective Attention Model for Target Speaker Extraction', 'meng24b_interspeech', 'time-domain configuration fasnet filter-and-sum si-sdr inter-channel separator two-speaker anechoic cocktail'], ['Liuxian Ma, Lin Shen, Ruobing Li, Haojie Zhang, Kun Qian, Bin Hu, Björn W. Schuller, Yoshiharu Yamamoto', 'E-ODN: An Emotion Open Deep Network for Generalised and Adaptive Speech Emotion Recognition', 'ma24_interspeech', 'emotional type infer ser model widest finer-grained uneven range unbalanced'], ['Hongyang Chen, Yuhong Yang, Zhongyuan Wang, Weiping Tu, Haojun Ai, Cedar Lin', 'Exploring Sentence Type Effects on the Lombard Effect and Intelligibility Enhancement: A Comparative Study of Natural and Grid Sentences', 'chen24j_interspeech', 'lct pronounced superior normal-to-lombard corpus valuable enhancing maintaining perspective focusing'], ['Masaya Kawamura, Ryuichi Yamamoto, Yuma Shirahata, Takuya Hasumi, Kentaro Tachibana', 'LibriTTS-P: A Corpus with Speaking Style and Speaker Identity Prompts for Text-to-Speech and Style Captioning', 'kawamura24_interspeech', 'libritts-r annotation prompt prompt-based tt speaker-level model dataset controllable conventional'], ['Yujie Chen, Jiangyan Yi, Jun Xue, Chenglong Wang, Xiaohui Zhang, Shunbo Dong, Siding Zeng, Jianhua Tao, Zhao Lv, Cunhang Fan', 'RawBMamba: End-to-End Bidirectional State Space Model for Audio Deepfake Detection', 'chen24k_interspeech', 'long-range fake mamba bonafide capture short information short-range sinc combining'], ['Xujiang Xing, Mingxing Xu, Thomas Fang Zheng', 'A Joint Noise Disentanglement and Adversarial Training Framework for Robust Speaker Verification', 'xing24_interspeech', 'noise-independent encoder embedding condition discourage supervise module speaker-invariant noisy space'], ['Chaeyoung Jung, Suyeon Lee, Ji-Hoon Kim, Joon Son Chung', 'FlowAVSE: Efficient Audio-Visual Speech Enhancement with Conditional Flow Matching', 'jung24b_interspeech', 'diffusion-based inference speed quality github reduces u-net output learnable degrading'], ['Yosuke Kashiwagi, Hayato Futami, Emiru Tsunoo, Siddhant Arora, Shinji Watanabe', 'Rapid Language Adaptation for Multilingual E2E Speech Recognition Using Encoder Prompting', 'kashiwagi24_interspeech', 'ctc language-specific prompt self-conditioned conditionally model zero-shot incoming attention-based multi-task'], ['Zhaoqing Li, Haoning Xu, Tianzi Wang, Shoukang Hu, Zengrui Jin, Shujie Hu, Jiajun Deng, Mingyu Cui, Mengzhe Geng, Xunying Liu', 'One-pass Multiple Conformer and Foundation Speech Systems Compression and Quantization Using An All-in-one Neural Model', 'li24o_interspeech', 'librispeech- wer switchboard- nested incurring single speed-up asr width store'], ['Emiru Tsunoo, Hayato Futami, Yosuke Kashiwagi, Siddhant Arora, Shinji Watanabe', 'Decoder-only Architecture for Streaming End-to-end Speech Recognition', 'tsunoo24_interspeech', 'blockwise prompt asr lm decoder subnetwork ample speech-processing test-other truncated'], ['Sichen Jin, Youngmoon Jung, Seungjin Lee, Jaeyoung Roh, Changwoo Han, Hoonyoung Cho', 'CTC-aligned Audio-Text Embedding for Streaming Open-vocabulary Keyword Spotting', 'jin24d_interspeech', 'kw text libriphrase frame non-streaming mere aligns aggregated attain on-the-fly'], ['Kiyoshi Kurihara, Masanori Sano', 'Enhancing Japanese Text-to-Speech Accuracy with a Novel Combination Transformer-BERT-based G2P: Integrating Pronunciation Dictionaries and Accent Sandhi', 'kurihara24_interspeech', 'transformer numeral counter bert transformer-based noun external proper dictionary employ'], ['Jiahao Li, Miao Liu, Shu Yang, Jing Wang, Xiang Xie', 'Motion Based Audio-Visual Segmentation', 'li24p_interspeech', 'av object task video mva module attention crossmodal pixel optical'], ['Le Xu, Jiangyan Yi, Tao Wang, Yong Ren, Rongxiu Zhong, Zhengqi Wen, Jianhua Tao', 'Residual Speaker Representation for One-Shot Voice Conversion', 'xu24b_interspeech', 'timbre robustness multi-layer approximation unseen encountering limited control challenge page'], ['Hayato Futami, Siddhant Arora, Yosuke Kashiwagi, Emiru Tsunoo, Shinji Watanabe', 'Finding Task-specific Subnetworks in Multi-task Spoken Language Understanding Model', 'futami24_interspeech', 'slu task forgetting previously trained adapting experiencing subnetwork mitigated catastrophic'], ['Pengfei Cai, Yan Song, Kang Li, Haoyu Song, Ian McLoughlin', 'MAT-SED: A Masked Audio Transformer with Masked-Reconstruction Based Pre-training for Sound Event Detection', 'cai24_interspeech', 'psds sed dcase network context pre-trained encoder global-local rnn-based positional'], ['Yuliya Korotkova, Ilya Kalinovskiy, Tatiana Vakhrusheva', 'Word-level Text Markup for Prosody Control in Speech Synthesis', 'korotkova24_interspeech', 'hand-labeling intonation tt one-to-many prosodic scalability interpretable expertise quantized problem'], ['Ryandhimas E. Zezario, Fei Chen, Chiou-Shann Fuh, Hsin-Min Wang, Yu Tsao', 'Non-Intrusive Speech Intelligibility Prediction for Hearing Aids using Whisper and Metadata', 'zezario24_interspeech', 'mbi-net clarity embeddings metric hearing-aid haspi top-performing validating pivotal intrusive'], ['Anfeng Xu, Kevin Huang, Tiantian Feng, Lue Shen, Helen Tager-Flusberg, Shrikanth Narayanan', 'Exploring Speech Foundation Models for Speaker Diarization in Child-Adult Dyadic Interactions', 'xu24c_interspeech', 'understanding child exemplary opened pathway demographic adopting vast opportunity addressing'], ['Mario Zusag, Laurin Wagner, Bernhad Thallinger', 'CrisperWhisper: Accurate Timestamps on Verbatim Speech Transcriptions', 'zusag24_interspeech', 'finetune hallucination tokenizer transcription timed cross-attention adjusting whisper filler adjustment'], ['Andrés Piñeiro-Martín, Carmen García-Mateo, Laura Docio-Fernandez, María del Carmen López-Pérez, Georg Rehm', 'Weighted Cross-entropy for Low-Resource Languages in Multilingual Speech Recognition', 'pineiromartin24_interspeech', 'high-resource wer whisper asr reduction continual model unbalanced fine-tune remarkable'], ['Yujie Yan, Xiran Xu, Haolin Zhu, Pei Tian, Zhongshu Ge, Xihong Wu, Jing Chen', 'Auditory Attention Decoding in Four-Talker Environment with EEG', 'yan24b_interspeech', 'aad asad trf cortical spatial indicated unattended two-talker scenario lateralization'], ['Cunhang Fan, Shunbo Dong, Jun Xue, Yujie Chen, Jiangyan Yi, Zhao Lv', 'Frequency-mix Knowledge Distillation for Fake Speech Detection', 'fan24_interspeech', 'fsd telephony asvspoof generalization scenario competitively combat undergoes model information'], ['Yafeng Chen, Siqi Zheng, Hui Wang, Luyao Cheng, Qian Chen, Shiliang Zhang, Junjie Li', 'ERes2NetV2: Boosting Short
11146-Duration Speaker Verification Performance with Computational Efficiency', 'chen24l_interspeech', 'trial net voxceleb fusion tasked sub-optimal feature multi-scale backbone ultimately'], ['Anika A. Spiesberger, Andreas Triantafyllopoulos, Alexander Kathan, Anastasia Semertzidou, Caterina Gawrilow, Tilman Reinelt, Wolfgang A. Rauch, Björn Schuller', '“So . . . my child . . . ” – How Child ADHD Influences the Way Parents Talk', 'spiesberger24_interspeech', 'mental symptomatology exerts parental parent-child surpassed well-being preschool cumbersome impractical'], ['Lifeng Zhou, Yuke Li, Rui Deng, Yuting Yang, Haoqi Zhu', 'Cross-Modal Denoising: A Novel Training Paradigm for Enhancing Speech-Image Retrieval', 'zhou24c_interspeech', 'modality framework inference feature alignment flickr interaction dataset effective task'], ['Malin Svensson Lundmark', 'Magnitude and timing of acceleration peaks in stressed and unstressed syllables', 'svenssonlundmark24_interspeech', 'velocity lower articulator lip peak measured segment posture expands accounted'], ['Ruizhe Huang, Mahsa Yarmohammadi, Sanjeev Khudanpur, Daniel Povey', 'Improving Neural Biasing for Contextual Speech Recognition by Early Context Injection and Text Perturbation', 'huang24f_interspeech', 'rare perturb inject technique enforce context-aware model asr merely encoders'], ['Yue Li, Xinsheng Wang, Li Zhang, Lei Xie', 'SCDNet: Self-supervised Learning Feature based Speaker Change Detection', 'li24q_interspeech', 'scd ssl wavlm wav vec model potent discern showcase task'], ['Zijie Lin, Tianyu He, Siqi Cai, Haizhou Li', 'ASA: An Auditory Spatial Attention Dataset with Multiple Speaking Locations', 'lin24f_interspeech', 'asad datasets localizing sound electroencephalography featuring cocktail attended source eeg'], ['Nan Chen, Yonghe Wang, Feilong Bao', 'Parameter-Efficient Adapter Based on Pre-trained Models for Speech Translation', 'chen24m_interspeech', 'peft parameter trainable fine-tuning parameter-sharing poorest non-negligible performance na lora'], ['Tin Mei Lun, Ekaterina Voskoboinik, Ragheb Al-Ghezi, Tamas Grosz, Mikko Kurimo', 'Oversampling, Augmentation and Curriculum Learning for Speaking Assessment with Limited Training Data', 'lun24_interspeech', 'finland imbalance greatest finnish evaluates swedish remarkable boost proficiency low-resource'], ['Fei Zhao, Jinjiang Liu, Xueliang Zhang', 'SDAEC: Signal Decoupling for Advancing Acoustic Echo Cancellation', 'zhao24b_interspeech', 'energy reference scaling subsequent microphone network impedes multiplied cancel factor'], ['YongKang Yin, Xu Li, Ying Shan, YueXian Zou', 'AFL-Net: Integrating Audio, Facial, and Lip Modalities with a Two-step Cross-attention for Robust Speaker Diarization in the Wild', 'yin24_interspeech', 'avr-net modality enhance obscured strategy fuse sufficiently multi-modal randomly scene'], ['Genshun Wan, Mengzhi Wang, Tingzhi Mao, Hang Chen, Zhongfu Ye', 'Lightweight Transducer Based on Frame-Level Criterion', 'wan24_interspeech', 'blank output probability achieving element encoder decoder truncate memory non-blank'], ['Hiroki Kanagawa, Takafumi Moriya, Yusuke Ijima', 'Pre-training Neural Transducer-based Streaming Voice Conversion for Faster Convergence and Alignment-free Training', 'kanagawa24_interspeech', 'vc-t seq guiding alignment tensor probable path pipeline matrix improbable'], ['Jing Wu, Ting Chen, Minchuan Chen, Wei Hu, Shaojun Wang, Jing Xiao', 'Improving Multilingual Text-to-Speech with Mixture-of-Language-Experts and Accent Disentanglement', 'wu24f_interspeech', 'code-switching tt conquer intra-utterance authenticity seamless comprehensibility aligns thorough fuse'], ['Jens Edlund, Christina Tånnander, Sébastien Le Maguer, Petra Wagner', 'Assessing the impact of contextual framing on subjective TTS quality', 'edlund24_interspeech', 'evaluation mo situation generalize without varying decontextualized information-rich fiction corroborates'], ['Kexin Wang, Carlos Ishi, Ryoko Hayashi', 'A multimodal analysis of different types of laughter expression in conversational dialogues', 'wang24r_interspeech', 'mirthful boosting gaze body facial softening tenser form predominant nonverbal'], ['Shunsuke Kando, Yusuke Miyao, Jason Naradowsky, Shinnosuke Takamichi', 'Textless Depen
11146dency Parsing by Labeled Sequence Prediction', 'kando24_interspeech', 'capturing method tree excels effectiveness acoustic speech cascading feature asr'], ['Zugang Zhao, Jinghong Zhang, Yonghui Liu, Jianbing Liu, Kai Niu, Zhiqiang He', 'Streamlining Speech Enhancement DNNs: an Automated Pruning Method Based on Dependency Graph with Advanced Regularized Loss Strategies', 'zhao24c_interspeech', 'compression unveils high-performing burgeoning quest layer impede enriches size computational'], ['Jingru Lin, Meng Ge, Junyi Ao, Liqun Deng, Haizhou Li', 'SA-WavLM: Speaker-Aware Self-Supervised Pre-training for Mixture Speech', 'lin24g_interspeech', 'quot pre-trained pipeline speaker shuffling model single-speaker merged limiting individually'], ['Ziyang Ma, Mingjie Chen, Hezhao Zhang, Zhisheng Zheng, Wenxi Chen, Xiquan Li, Jiaxin Ye, Xie Chen, Thomas Hain', 'EmoBox: Multilingual Multi-corpus Speech Emotion Recognition Toolkit and Benchmark', 'ma24b_interspeech', 'ser intra-corpus cross-corpus datasets setting balanced fully making suffered language'], ['Haoyu Li, Baochen Yang, Yu Xi, Linfeng Yu, Tian Tan, Hao Li, Kai Yu', 'Text-aware Speech Separation for Multi-talker Keyword Spotting', 'li24r_interspeech', 'kw clue mixed permutation greatly keyword-specific determinization noisy cocktail front-ends'], ['Jaesong Lee, Soyoon Kim, Hanbyul Kim, Joon Son Chung', 'Lightweight Audio Segmentation for Long-form Speech Translation', 'lee24h_interspeech', 'model quality consume overall segment partitioned pre-training system exists self-supervised'], ['Martin Lebourdais, Théo Mariotte, Antonio Almudévar, Marie Tahon, Alfonso Ortega', 'Explainable by-design Audio Segmentation through Non-Negative Matrix Factorization and Probing', 'lebourdais24_interspeech', 'interpretable good representation explanation latent modularity informativeness forensics property compactness'], ['Nan Zhou, Youhai Jiang, Jialin Tan, Chongmin Qi', 'PLDNet: PLD-Guided Lightweight Deep Network Boosted by Efficient Attention for Handheld Dual-Microphone Speech Enhancement', 'zhou24d_interspeech', 'u-net mobile low-complexity guidance phone algorithm pre-process top-performing era gated'], ['Martina Valente, Fabio Brugnara, Giovanni Morrone, Enrico Zovato, Leonardo Badino', 'Exploring Spoken Language Identification Strategies for Automatic Transcription of Multilingual Broadcast and Institutional Speech', 'valente24_interspeech', 'diarization sli monolingual reduction relative wer lower change negatively observing'], ['Kenichi Fujita, Takanori Ashihara, Marc Delcroix, Yusuke Ijima', 'Lightweight Zero-shot Text-to-Speech with Mixture of Adapters', 'fujita24b_interspeech', 'tt method speaker module reproducing adapter non-autoregressive characteristic fidelity less'], ['Junzhe Liu, Jianwei Yu, Xie Chen', 'Improved Factorized Neural Transducer Model For Text-only Domain Adaptation', 'liu24e_interspeech', 'ifnt fnt wer datasets cer out-of-domain relative compared vocabulary doubt'], ['Jakub Hoscilowicz, Adam Wiacek, Jan Chojnacki, Adam Cieslak, Leszek Michon, Artur Janicki', 'Non-Linear Inference Time Intervention: Improving LLM Truthfulness', 'hoscilowicz24_interspeech', 'iti truthful multiple-choice probing pointing let manifest forest relative truth'], ['Benjamin Elie, Juraj Simko, Alice Turk', 'A data-driven model of acoustic speech intelligibility for optimization-based models of speech production', 'elie24_interspeech', 'hyper-articulation articulatory least bilstm-based effort hypo lindblom phoneme balancing return'], ['Jinghong Zhang, Zugang Zhao, Yonghui Liu, Jianbing Liu, Zhiqiang He, Kai Niu', 'TD-PLC: A Semantic-Aware Speech Encoding for Improved Packet Loss Concealment', 'zhang24g_interspeech', 'plc facilitating integrates transmission audio semantic scenario integrity prolonged concurrently'], ['Antonio Almudévar, Théo Mariotte, Alfonso Ortega, Marie Tahon, Luis Vicente, Antonio Miguel, Eduardo Lleida', 'Predefined Prototypes for Intra-Class Separation and Disentanglement', 'almudevar24_interspeech', 'embeddings explainable advantage class inter-class disentangling separability different prototypical simplify'], ['Tina Raissi, Christoph Lüscher, Simon Berger, Ralf Schlüter, Hermann Ney', 'Investigating the Effect of Label Topology and Training Criterion on ASR Performance and Alignment Quality', 'raissi24_interspeech', 'comparison hmm dis factored monotonic first-order strictly division classic modular'], ['Kira Tulchynska, Sylvanus Job, Alena Witzlack-Makarevich, Margaret Zellers', 'Prosodic marking of syntactic boun
11146daries in Khoekhoe', 'tulchynska24_interspeech', 'clause-final auxiliary clause position lengthening candidate intonation sov naq following'], ['Premanand Nayak, Kamini Sabu, M. Ali Basha Shaik', 'Multi-mic Echo Cancellation Coalesced with Beamforming for Real World Adverse Acoustic Conditions', 'nayak24_interspeech', 'aec mmaec deep beamformer ser erle psd introduce far-end steering'], ['Martina Di Bratto, Maria Di Maro, Antonio Origlia', 'On the Use of Plausible Arguments in Explainable Conversational AI', 'dibratto24_interspeech', 'recommendation usability concerning delf cross-disciplinary argumentative dialogue believable system interaction'], ['Yaoyao Yue, Michael Proctor, Luping Zhou, Rijul Gupta, Tharinda Piyadasa, Amelia Gully, Kirrie Ballard, Craig Jin', 'Towards Speech Classification from Acoustic and Vocal Tract data in Real-time MRI', 'yue24_interspeech', 'multimodal rtmri unimodal transformer audio phonemic stream model articulatory combine'], ['Thanapat Trachu, Chawan Piansaddhayanon, Ekapol Chuangsuwanich', 'Thunder : Unified Regression-Diffusion Speech Enhancement with a Single Reverse Step using Brownian Bridge', 'trachu24_interspeech', 'diffusion model mode regression score-based necessitate regression-based initializing diffusion-based instability'], ['Rohit Paturi, Xiang Li, Sundararajan Srinivasan', 'AG-LSEC: Audio Grounded Lexical Speaker Error Correction', 'paturi24_interspeech', 'wder audio-based diarization pipeline asr system beat callhome relative fisher'], ['Sheng-Chieh Chiu, Chia-Hua Wu, Jih-Kang Hsieh, Yu Tsao, Hsin-Min Wang', 'Learnable Layer Selection and Model Fusion for Speech Self-Supervised Learning Models', 'chiu24_interspeech', 'gumbel ssl dimension-wise fusing interleaved superb promise enhances sum concatenation'], ['Yicong Jiang, Tianzi Wang, Xurong Xie, Juan Liu, Wei Sun, Nan Yan, Hui Chen, Lan Wang, Xunying Liu, Feng Tian', 'Perceiver-Prompt: Flexible Speaker Adaptation in Whisper for Chinese Disordered Speech Recognition', 'jiang24b_interspeech', 'dysarthric afflicted non-dysarthric lora profound stemming perceiver fixed-length variable-length dissimilarity'], ['Premanand Nayak, M. Ali Basha Shaik', 'Elucidating Clock-drift Using Real-world Audios In Wireless Mode For Time-offset Insensitive End-to-End Asynchronous Acoustic Echo Cancellation', 'nayak24b_interspeech', 'non-linear exacerbation device delineate clock cancellers revisit conventional playback causing'], ['Hao Tan, Xiaochen Liu, Huan Zhang, Junjian Zhang, Yaguan Qian, Zhaoquan Gu', 'DualPure: An Efficient Adversarial Purification Method for Speech Command Recognition', 'tan24_interspeech', 'purify malicious defense perturbation attack frequency unconditional disrupt white-box black-box'], ['Robert Flynn, Anton Ragni', 'Self-Train Before You Transcribe', 'flynn24_interspeech', 'self-training adaptation domain teacher student gain test-time training noisy utilises'], ['Bulat Khaertdinov, Pedro Jeruis, Annanda Sousa, Enrique Hortal', 'Exploring Self-Supervised Multi-view Contrastive Learning for Speech Emotion Recognition with Limited Annotations', 'khaertdinov24_interspeech', 'ser ssl unprecedented reaching performance costly unweighted paralinguistic pre-training advancement'], ['Chengxu Yang, Lin Zheng, Sanli Tian, Gaofeng Cheng, Sujie Xiao, Ta Li', 'Contextual Biasing with Confidence-based Homophone Detector for Mandarin End-to-End Speech Recognition', 'yang24j_interspeech', 'fusion shallow phrase method infrequently deep asr aishell- incorrectly relative'], ['Robert Flynn, Anton Ragni', 'How Much Context Does My Attention-Based ASR System Need?', 'flynn24b_interspeech', 'second long-format length tedlium uncommon acoustic podcasts positional width zero-shot'], ['Sophie Fagniart, Brigitte Charlier, Véronique Delvaux, Bernard Harmegnies, Anne Huberlant, Myriam Piccaluga, Kathy Huet', 'Production of fricative consonants in French-speaking children with cochlear implants and typical hearing: acoustic and phonological analyses.', 'fagniart24_interspeech', 'group stopping lower percentage higher amplitude correct mid-frequency chronological value'], ['Ming Gao, Hang Chen, Jun Du, Xin Xu, Hongxiao Guo, Hui Bu, Jianxing Yang, Ming Li, Chin-Hui Lee', 'Enhancing Voice Wake-Up for Dysarthria: Mandarin Dysarthria Speech Corpus Release and Customized System Design', 'gao24c_interspeech', 'mdsc 
11146wws dysarthric home individual intelligibility effortless exceptional encompasses challenge'], ['Jeremy Chang, Kuan-Yu Chen, Chung-Hsien Wu', 'Applying Reinforcement Learning and Multi-Generators for Stage Transition in an Emotional Support Dialogue System', 'chang24_interspeech', 'rouge-l empathetic distress bleu skill user experiencing response grown coping'], ['Bingsong Bai, Fengping Wang, Yingming Gao, Ya Li', 'SPA-SVC: Self-supervised Pitch Augmentation for Singing Voice Conversion', 'bai24_interspeech', 'svc cross-domain scenario model innovatively ssim posing diffusion-based disparity hoarseness'], ['Song Li, Yongbin You, Xuezhi Wang, Zhengkun Tian, Ke Ding, Guanglu Wan', 'MSR-86K: An Evolving, Multilingual Corpus with 86,300 Hours of Transcribed Audio for Speech Recognition Research', 'li24s_interspeech', 'asr whisper publicly impeded gateway chatgpt huggingface garnered immense pave'], ['Zheshu Song, Jianheng Zhuo, Yifan Yang, Ziyang Ma, Shixiong Zhang, Xie Chen', 'LoRA-Whisper: Parameter-Efficient and Extensible Multilingual ASR', 'song24_interspeech', 'language lora interference persist witnessed mitigating emergence degrading performance incorporation'], ['Na Hu, Hugo Schnack, Amalia Arvaniti', 'Automatic pitch accent classification through image classification', 'hu24b_interspeech', 'classifying showcasing effectiveness feature pixel posed feature-based type labelling yielded'], ['Wei-lin Xie, Yu-Xuan Xi, Yan Song, Jian-tao Zhang, Hao-yu Song, Ian McLoughlin', 'DB-PMAE: Dual-Branch Prototypical Masked AutoEncoder with locality for domain robust speaker verification', 'xie24b_interspeech', 'sid pretrained cnceleb similarity siamese finetuned finetuning frontend framework remarkably'], ['Dan Oneata, Herman Kamper', 'Translating speech with just images', 'oneata24_interspeech', 'translation caption low-resource audio image yoruba captioning regime albeit grounded'], ['Per E Kummervold, Javier de la Rosa, Freddy Wetjen, Rolv-Arild Braaten, Per Erik Solberg', 'Whispering in Norwegian: Navigating Orthographic and Dialectic Challenges', 'kummervold24_interspeech', 'openai whisper dataset large-v fleurs summarise weakly translating converting tailored'], ['David Ortiz-Perez, Jose Garcia-Rodriguez, David Tomás', 'Cognitive Insights Across Languages: Enhancing Multimodal Interview Analysis', 'ortizperez24_interspeech', 'decline taukadial initiating anomalous in-depth audio mild transcribe differentiate comprises'], ['Théo Mariotte, Anthony Larcher, Silvio Montrésor, Jean-Hugh Thomas', 'ASoBO: Attentive Beamformer Selection for Distant Speaker Diarization in Meetings', 'mariotte24_interspeech', 'osd vad spatial self-attention-based filter steer speech-processing explainability multi-microphone segment'], ['Rong Gong, Hongfei Xue, Lezhi Wang, Xin Xu, Qisheng Li, Lei Xie, Hui Bu, Shaomei Wu, Jiaming Zhou, Yong Qin, Binbin Zhang, Jun Du, Jia Bin, Ming Li', 'AS-70: A Mandarin stuttered speech dataset for automatic speech recognition and stuttering event detection', 'gong24_interspeech', 'asr inclusivity diminishes human-level verbatim sed encompassing task speech-related stand'], ['Sumit Ranjan, Rupayan Chakraborty, Sunil Kumar Kopparapu', 'Reinforcement Learning based Data Augmentation for Noise Robust Speech Emotion Recognition', 'ranjan24_interspeech', 'ser technique building empathetic scenario picking machine validating cross-corpus reward'], ['Bilal Rahou, Hervé Bredin', 'Multi-latency look-ahead for streaming speaker segmentation', 'rahou24_interspeech', 'latency frame-wise detection inference prism bare contribution computational causal overlapped'], ['Hiroki Kanagawa, Yusuke Ijima', 'Knowledge Distillation from Self-Supervised Representation Learning Model with Discrete Speech Units for Any-to-Any Streaming Voice Conversion', 'kanagawa24b_interspeech', 'ssl content teacher offline prosody operation student encoder inferencing vcs'], ['Yuchen Hu, Chen Chen, Ruizhe Li, Qiushi Zhu, Eng Siong Chng', 'Noise-aware Speech Enhancement using Diffusion Probabilistic Model', 'hu24c_interspeech', 'nase noise reverse guide unseen plug-and-play surge mainstream attracted specificity'], ['Jinming Chen, Jingyi Fang, Yuanzhong Zheng, Yaoxuan Wang, Haojun Fei', 'Qifusion-Net: Layer-adapted Stream/Non-stream Model for End-to-End Multi-Accent Speech Recognition', 'chen24n_interspeech', 'fusion accent auto facilitating multi chunk streaming cer fine-grained frame-level'], ['Tianci Wu, Shulin He, Jiahui Pan, Haifeng Huang, Zhijian Mo, Xueliang Zhang', 'Unified Audio Visual Cues for Target Speaker Extraction', 'wu24h_interspeech', 'lip fuse distinct network divide-and-conquer movement capitalizing attention tse blend'], ['Hui Wang, Shiwan Zhao, Jiaming Zhou, Xiguang Zheng, Haoqin Sun, Xuechen Wang, Yong Qin', 'Uncertainty-Aware Mean Opinion Score Prediction', 'wang24s_interspeech', 'mo uncertainty diverse system practical epistemic heteroscedastic hindering real monte'], ['Marcely Zanon Boito, Vivek Iyer, Nikolaos Lagos, Laurent Besacier, Ioan Calapodescu', 'mHuBERT-147: A Compact Multilingual HuBERT Model', 'zanonboito24_interspeech', 'params hour larger ml-superb up-sampling leaderboards mm competitiveness massively unprecedented'], ['Seung-bin Kim, Chan-yeong Lim, Jungwoo Heo, Ju-ho Kim, Hyun-seo Shin, Kyo-Won K
11146oo, Ha-Jin Yu', 'MR-RawNet: Speaker verification system with multiple temporal resolutions for variable duration utterances using raw waveforms', 'kim24j_interspeech', 'multi-resolution waveform-based robustness adjusts persistent obstacle optimally utilization ensuring insufficient'], ['Dejan Porjazovski, Anssi Moisio, Mikko Kurimo', 'Out-of-distribution generalisation in spoken language understanding', 'porjazovski24_interspeech', 'ood slu split slurp modified data emphasising technique unexpectedly dataset'], ['Juan M. Martín-Doñas, Aitor Álvarez, Eros Rosello, Angel M. Gomez, Antonio M. Peinado', 'Exploring Self-supervised Embeddings and Synthetic Data Augmentation for Robust Audio Deepfake Detection', 'martindonas24_interspeech', 'classifier downstream upstream revisit complemented anti-spoofing feed augmenting specialized foundation'], ['Kou Tanaka, Hirokazu Kameoka, Takuhiro Kaneko, Yuto Kondo', 'PRVAE-VC2: Non-Parallel Voice Conversion by Distillation of Speech Representations', 'tanaka24_interspeech', 'streamable hubert converted gap hidden-unit persists webpage posing many-to-many non-streaming'], ['Markéta Řezáčková, Daniel Tihelka, Jindřich Matoušek', 'Homograph Disambiguation with Text-to-Text Transfer Transformer', 'rezackova24_interspeech', 'single-model wikipedia grapheme-to-phoneme fine-tuned model published proved outperformed powerful online'], ['Séverine Guillaume, Maxime Fily, Alexis Michaud, Guillaume Wisniewski', 'Gender and Language Identification in Multilingual Models of Speech: Exploring the Genericity and Robustness of Speech Representations', 'guillaume24_interspeech', 'xls-r unispeech snippet distill assess informational audio away irrelevant refined'], ['Beomseok Lee, Ioan Calapodescu, Marco Gaido, Matteo Negri, Laurent Besacier', 'Speech-MASSIVE: A Multilingual Speech Dataset for SLU and Beyond', 'lee24i_interspeech', 'massive language datasets slot-filling inherits massively few-shot assess versatile benchmarking'], ['Longbiao Cheng, Ashutosh Pandey, Buye Xu, Tobi Delbruck, Shih-Chii Liu', 'Dynamic Gated Recurrent Neural Network for Compute-efficient Speech Enhancement', 'cheng24_interspeech', 'rnn gru gate select step model resource-constrained rnn-based dns neuron'], ['Félix Saget, Meysam Shamsi, Marie Tahon', 'Lifelong Learning MOS Prediction for Synthetic Speech Quality Evaluation', 'saget24_interspeech', 'predictor mode bvcc chronological long-standing breakthrough cross-corpus synthesis reproducible blizzard'], ['Mark Huckvale, Gaston Hilkhuysen', 'Evaluating a 3-factor listener model for prediction of speech intelligibility to hearing-impaired listeners', 'huckvale24_interspeech', 'pure-tone impaired hearing factor sensitivity threshold listener-dependent audiogram held-out evaluate'], ['Séverin Baroudi, Thomas Pellegrini, Hervé Bredin', 'Specializing Self-Supervised Speech Representations for Speaker Segmentation', 'baroudi24_interspeech', 'diarization pretraining wavlm single-speaker pretrained multi-speaker conversational specialize pretext pre-segmented'], ['Xun Gong, Anqi Lv, Zhiming Wang, Yanmin Qian', 'Contextual Biasing Speech Recognition in Speech-enhanced Large Language Model', 'gong24b_interspeech', 'speechllm bias wer persists hallucination declined test-clean word reduction relative'], ['Yangze Li, Xiong Wang, Songjun Cao, Yike Zhang, Long Ma, Lei Xie', 'A Transcription Prompt-based Efficient Audio Large Language Model for Robust Speech Recognition', 'li24t_interspeech', 'repetition decoding llm k-hour nar tokenizer illusion fundamentally non-autoregressive cer'], ['Cheng Gong, Erica Cooper, Xin Wang, Chunyu Qiang, Mengzhe Geng, Dan Wells, Longbiao Wang, Jianwu Dang, Marc Tessier, Aidan Pine, Korin Richmond, Junichi Yamagishi', 'An Initial Investigation of Language Adaptation for TTS Systems under Low-resource Scenarios', 'gong24c_interspeech', 'fine-tuning multilingual similarity ssl-based massively adaptability data target surprisingly audio-only'], ['Lingwei Meng, Jiawen Kang, Yuejiao Wang, Zengrui Jin, Xixin Wu, Xunying Liu, Helen Meng', 'Empowering Whisper as a Joint Multi-Talker and Target-Talker Speech Recognition System', 'meng24c_interspeech', 'talker task embedding librispeechmix librimix empower plug freeze pioneering separator'], ['Takuto Igarashi, Yuki Saito, Kentaro Seki, Shinnosuke Takamichi, Ryuichi Yamamoto, Kentaro Tachibana, Hiroshi Saruwatari', 'Noise-Robust Voice Conversion by Conditional Denoising Training Using Latent Variables of Recording Quality and Environment', 'igarashi24_interspeech', 'speech converted source noisy-to-clean utterance-wise conventional frame-wise noise scene model'], ['Zuoliang Li, Wu Guo, Bin Gu, Shengyu Peng, Jie Zhang', 'Contrastive Learning and Inter-Speaker Distribution Alignment Based Unsupervised Domain Adaptation for Robust Speaker Verification', 'li24u_interspeech', 'uda source target source-domain target-domain method cn-celeb momentum compactness mitigating'], ['Kubilay Can Demir, Belén Lojo Rodríguez, Tobias Weise, Andreas Maier, Seung Hee Yang', 'Towards Intelligent Speech Assistants in Operating Rooms: A Multimodal Model for Surgical Workflow Analysis', 'demir24_interspeech', 'phase operation mono-modal merges seamlessly prerequisite multi-stage frame-wise -score gated'], ['Dong-Hyun Kim, Joon-Hyuk Chang', 'Mitigating Overfitting in Structured Pruning of ASR Models with Gradient-Guided Parameter Regularization', 'kim24k_interspeech', 'availability shift model exacerbated confront limited generality regularize catastrophic forgetting'], ['Oliver Schrüfer, Manuel Milling, Felix Burkhardt, Florian Eyben, Björn Schuller', 'Are you sure? Analysing Uncertainty Quantification Approaches for Real-world Speech Emotion Recognition', 'schrufer24_interspeech', 'ser faulty ood prediction ambiguity reliable out-of-distribution many documented indication'], ['Katharina Anderer, Andreas Reich, Matthias Wölfel', 'MaViLS, a Benchmark Dataset for Video-to-Slide Alignment, Assessing Baseline Accuracy with a Multimodal Alignment Algorithm Leveraging Speech, OCR, and Visual Features', 'anderer24_interspeech', 'slide lecture highlight matching image video sift underscoring challenge optical'], ['Jinlong Xue, Yayue Deng, Yingming Gao, Ya Li', 'Retrieval Augmented Generation in Prompt-based Text-to-Speech Synthesis with Context-Aware Contrastive Language-Audio Pretraining', 'xue24b_interspeech', 'prompt rag tt llm speech style-related in-context akin clone model'], ['Gábor Gosztoly
11146a, László Tóth', "Combining Acoustic Feature Sets for Detecting Mild Cognitive Impairment in the Interspeech'24 TAUKADIAL Challenge", 'gosztolya24_interspeech', 'mci task cross-validation machine workflow functionals custom learning mmse rmse'], ['Zhiyuan Tang, Dong Wang, Shen Huang, Shidong Shang', 'Pinyin Regularization in Error Correction for Chinese Speech Recognition with Large Language Models', 'tang24_interspeech', 'llm dataset error-correcting paradise hypothesis asr website specialized straightforward enhances'], ['Maryam Naderi, Enno Hermann, Alexandre Nanchen, Sevada Hovsepyan, Mathew Magimai.-Doss', 'Towards interfacing large language models with ASR systems using confidence measures and prompting', 'naderi24_interspeech', 'llm transcript confidence-based post-hoc grow less rescoring n-best correction beyond'], ['Delphine Charuau, Andrea Briglia, Erika Godde, Gérard Bailly', 'Training speech-breathing coordination in computer-assisted reading', 'charuau24_interspeech', 'highlighting respiratory untrained assistance aloud repeated text principle trained karaoke'], ['Koichi Miyazaki, Yoshiki Masuyama, Masato Murata', 'Exploring the Capability of Mamba in Speech Applications', 'miyazaki24_interspeech', 'transformer-based ssms model e-branchformer evaluation long-form facto well-designed asr summarization'], ['Gábor Gosztolya, Mercedes Vetráb, Veronika Svindt, Judit Bóna, Ildikó Hoffmann', 'Wav2vec 2.0 Embeddings Are No Swiss Army Knife -- A Case Study for Multiple Sclerosis', 'gosztolya24b_interspeech', 'functionals surprisingly feature self-supervised classification model unannotated ecapa-tdnn audio performance'], ['Stefan Kalabakov, Monica Gonzalez-Machorro, Florian Eyben, Björn W. Schuller, Bert Arnrich', 'A Comparative Analysis of Federated Learning for Speech-Based Cognitive Decline Detection', 'kalabakov24_interspeech', 'healthy centralised solution hampered collectively discern alzheimer timely state healthcare'], ['Tim Polzehl, Tim Herzig, Friedrich Wicke, Kathleen Wermke, Razieh Khamsehashari, Michiko Dahlem, Sebastian Möller', 'Towards Classifying Mother Tongue from Infant Cries - Findings Substantiating Prenatal Learning Theory', 'polzehl24_interspeech', 'panns neonate model network adhere substantiate confounding cry thereof held-out'], ['Jérémy Giroud, Jessica Lei, Kirsty Phillips, Matthew H. Davis', 'Behavioral evidence for higher speech rate convergence following natural than artificial time altered speech', 'giroud24_interspeech', 'interaction artificially crucial phenomenon ai-powered multiplying proliferation amplified inform everyday'], ['Andrei Andrusenko, Aleksandr Laptev, Vladimir Bataev, Vitaly Lavrukhin, Boris Ginsburg', 'Fast Context-Biasing for CTC and Transducer ASR models with CTC-based Word Spotter', 'andrusenko24_interspeech', 'recognition candidate pressing nemo method nvidia contextualized model reuse slowing'], ['Huihang Zhong, Yanlu Xie, ZiJin Yao', 'Leveraging Large Language Models to Refine Automatic Feedback Generation at Articulatory Level in Computer Aided Pronunciation Training', 'zhong24b_interspeech', 'gpt- llm helpfulness invite potential effectiveness generated comprehensibility computer-aided capt'], ['Qiuming Zhao, Guangzhi Sun, Chao Zhang, Mingxing Xu, Thomas Fang Zheng', 'SAML: Speaker Adaptive Mixture of LoRA Experts for End-to-End ASR', 'zhao24d_interspeech', 'moe quantised model mixture-of-experts test-time adaptation personalised ted-lium resource-constrained conformer-based'], ['Siyang Wang, Éva Székely, Joakim Gustafson', 'Contextual Interactive Evaluation of TTS Models in Dialogue Systems', 'wang24t_interspeech', 'mo shortcoming setup context mean-opinion-score custom-built paramount system questioned evaluate'], ['Gábor Gosztolya, Veronika Svindt, Judit Bóna, Ildikó Hoffmann', 'Automatic Longitudinal Investigation of Multiple Sclerosis Subjects', 'gosztolya24c_interspeech', 'year healthy third category nervous control chronic workflow change classification'], ['Imen Ben-Amor, Jean-Francois Bonastre, Salima Mdhaffar', 'Extraction of interpretable and shared speaker-specific speech attributes through binary auto-encoder', 'benamor24_interspeech', 'explainability proposal attribute embeddings eer lack binarization attribute-based restructuring posing'], ['Anisia Popescu, Lori Lamel, Ioana Vasilescu, Laurence Devillers', 'Automatic Speech Recognition with parallel L1 and L2 acoustic phone models to evaluate /l/ allophony in L2 English speech production', 'popescu24_interspeech', 'french lateral staple darkness dar
11146k clearer asr documented less consuming'], ['Zizhen Lin, Xiaoting Chen, Junyu Wang', 'MUSE: Flexible Voiceprint Receptive Fields and Multi-Path Fusion Enhanced Taylor Transformer for U-Net-based Speech Enhancement', 'lin24h_interspeech', 'met u-net lightweight spatial csa deformable voicebank channel attention mere'], ['Yin-Long Liu, Rui Feng, Jia-Hong Yuan, Zhen-Hua Ling', "Clever Hans Effect Found in Automatic Detection of Alzheimer's Disease through Speech", 'liu24f_interspeech', 'pitt recording bias datasets preprocessed uncover audio necessity emphasize accessible'], ['Biswajit Karan, Joshua Jansen van Vüren, Febe de Wet, Thomas Niesler', 'A Transformer-Based Voice Activity Detector', 'karan24_interspeech', 'vad architecture multilingual benchmark end-to-end widens enterprise dataset off-the-shelf auc'], ['Tasnima Sadekova, Mikhail Kudinov, Vadim Popov, Assel Yermekova, Artem Khrapov', 'PitchFlow: adding pitch control to a Flow-matching based TTS model', 'sadekova24_interspeech', 'guidance generation recent diffusion high timbre denoising quality stability fine-grained'], ['Tonmoy Rajkhowa, Amartya Roy Chowdhury, Sankalp Nagaonkar, Achyut Mani Tripathi, Mahadeva Prasanna', 'TM-PATHVQA: 90000+ Textless Multilingual Questions for Medical Visual Question Answering', 'rajkhowa24_interspeech', 'vqa pathological speech-based dataset performing image scenario diagnostics system interaction'], ['Erfan A. Shams, Iona Gessinger, Patrick Cormac English, Julie Carson-Berndsen', 'Are Articulatory Feature Overlaps Shrouded in Speech Embeddings?', 'shams24_interspeech', 'probe assimilation activation transformer articulation explanatory probing spreading ipa phonetic'], ['Sevada Hovsepyan, Mathew Magimai.-Doss', "Neurocomputational model of speech recognition for pathological speech detection: a case study on Parkinson's disease speech detection", 'hovsepyan24_interspeech', 'healthy accumulates computational two-level uncover syllable auc modest merit plausible'], ['Anurag Chowdhur
11146y, Abhinav Misra, Mark C. Fuhs, Monika Woszczyna', 'Investigating Confidence Estimation Measures for Speaker Diarization', 'chowdhury24_interspeech', 'score system downstream identity error derived propagate speaker-adapted adversely segment'], ['Mengjie Qian, Siyuan Tang, Rao Ma, Kate M. Knill, Mark J.F. Gales', "Learn and Don't Forget: Adding a New Language to ASR Foundation Models", 'qian24_interspeech', 'ewc tuning soft fine-tuning code parameter capability performance consolidation set'], ['Benjamin van Niekerk, Julian Zaïdi, Marc-André Carbonneau, Herman Kamper', 'Spoken-Term Discovery using Discrete Speech Units', 'vanniekerk24_interspeech', 'pattern find sub-sequences longstanding discretization zerospeech zero-resource revisit challenge discovering'], ['Mukhtar Mohamed, Oli Danyi Liu, Hao Tang, Sharon Goldwater', 'Orthogonality and isotropy of speaker and phonetic information in self-supervised speech representations', 'mohamed24_interspeech', 'probing correlate centroid property downstream degree phone space spanned nuanced'], ['Jinlong Xue, Yayue Deng, Yicheng Han, Yingming Gao, Ya Li', 'Improving Audio Codec-based Zero-Shot Text-to-Speech Synthesis with Multi-Modal Context and Large Language Model', 'xue24c_interspeech', 'tt llm prompt leverage token adapt scenario audiobook -second personalized'], ['Alexis Plaquet, Hervé Bredin', 'On the calibration of powerset speaker diarization models', 'plaquet24_interspeech', 'low-confidence region confidence formulation random explore multiclass model unannotated validating'], ['Kishan Gupta, Nicola Pia, Srikanth Korse, Andreas Brendel, Guillaume Fuchs, Markus Multrus', 'On Improving Error Resilience of Neural End-to-End Speech Coders', 'gupta24c_interspeech', 'packet fec plc bitrates loss low resilient robustness like concealment'], ['Ryo Masumura, Naoki Makishima, Tomohiro Tanaka, Mana Ihori, Naotaka Kawata, Shota Orihashi, Kazutoshi Shinoda, Taiga Yamane, Saki Mizuno, Keita Suzuki, Satoshi Suzuki, Nobukatsu Hojo, Takafumi Moriya, Atsushi Ando', 'Unified Multi-Talker ASR with and without Target-speaker Enrollment', 'masumura24_interspeech', 'mt-asr modeling process form independent bridging enrolled mutually trained autoregressive'], ['Guinan Li, Jiajun Deng, Youjun Chen, Mengzhe Geng, Shujie Hu, Zhe Li, Zengrui Jin, Tianzi Wang, Xurong Xie, Helen Meng, Xunying Liu', 'Joint Speaker Features Learning for Audio-visual Multichannel Speech Separation and Recognition', 'li24v_interspeech', 'wavlm -ted feature purpose-built lr best-performing ecapa-tdnn tightly dev zero-shot'], ['Zexu Pan, Gordon Wichern, François G. Germain, Kohei Saijo, Jonathan Le Roux', 'PARIS: Pseudo-AutoRegressIve Siamese Training for Online Speech Separation', 'pan24_interspeech', 'streaming offline autoregressive step training-inference empowering network si-snr forcing regime'], ['Jizhen Li, Xinmeng Xu, Weiping Tu, Yuhong Yang, Rong Zhu', 'Improving Speech Enhancement by Integrating Inter-Channel and Band Features with Dual-branch Conformer', 'li24w_interspeech', 'channel t-f aware correlation time-frequency information dns-challenge relation recent different'], ['Vasileios Moschopoulos, Thanasis Kotsiopoulos, Pablo Peso Parada, Konstantinos Nikiforidis, Alexandros Stergiadis, Gerasimos Papakostas, Md Asif Jalal, Jisi Zhang, Anastasios Drosou, Karthikeyan Saravanan', 'Exploring compressibility of transformer based text-to-music (TTM) models', 'moschopoulos24_interspeech', 'params compression generative fad infeasible desktop state-of-the distillation server applicability'], ['Thomas Muller, Stephane Ragot, Laetitia Gros, Pierrick Philippe, Pascal Scalart', 'Speech quality evaluation of neural audio codecs', 'muller24c_interspeech', 'codec ev encodec dmos opus lpcnet bitrates targeting technological complement'], ['Yi-Cheng Lin, Haibin Wu, Huang-Cheng Chou, Chi-Chun Lee, Hung-yi Lee', 'Emo-bias: A Large Scale Evaluation of Social Bias on Speech Emotion Recognition', 'lin24i_interspeech', 'ser gender upstream model ssl toward exhibit cutting-edge aiding ssl-based'], ['Judith Dineley, Ewan Carr, Lauren L. White, Catriona Lucas, Zahia Rahman, Tian Pan, Faith Matcham, Johnny Downs, Richard J. Dobson, Thomas F. Quatieri, Nicholas Cummins', 'Variability of speech timing features across repeated recordings: a comparison of open-source extraction techniques', 'dineley24_interspeech', 'extracted clinical via replication hinder variation within-speaker susceptible week symptom'], ['Mahdi Amiri, Ina Kodrasi', 'Adversarial Robustness Analysis in Automatic Pathological Speech Detection Approaches', 'amiri24_interspeech', 'dl-based perturbation imperceptibility imperceptible ineffective healthcare vulnerability unexplored attend projected'], ['Chun-Yi 
11146Kuan, Wei-Ping Huang, Hung-yi Lee', 'Understanding Sounds, Missing the Questions: The Challenge of Object Hallucination in Large Audio-Language Models', 'kuan24_interspeech', 'lalms discriminative audio enhance overlooking assess captioning inadequate struggle weakness'], ['Yiyuan Yang, Niki Trigoni, Andrew Markham', 'Pre-training Feature Guided Diffusion Model for Speech Enhancement', 'yang24k_interspeech', 'efficiency streamline vae optimizes clarity deterministic reverse guidance pretraining utilization'], ['Chung-Wen Wu, Berlin Chen', 'Optimizing Automatic Speech Assessment: W-RankSim Regularization and Hybrid Feature Fusion Strategies', 'wu24i_interspeech', 'asa ordinal vector weighted handcrafted ssl closer showcasing imbalanced challenge'], ['Livia Qian, Gabriel Skantze', 'Joint Learning of Context and Feedback Embeddings in Spoken Dialogue', 'qian24b_interspeech', 'response appropriateness ranking conversational short neglecting backchannels function primarily carry'], ['Nicolò Loddo, Francisca Pessanha, Almila Akdag', 'What if HAL breathed? Enhancing Empathy in Human-AI Interactions with Breathing Speech Synthesis', 'loddo24_interspeech', 'agent novelty towards synthesized diverges empathetic dilemma deepen methodologically engagement'], ['Sameer Khurana, Chiori Hori, Antoine Laurent, Gordon Wichern, Jonathan Le Roux', 'ZeroST: Zero-Shot Speech Translation', 'khurana24_interspeech', 'foundation model multilingual mm bypassing synergistic massively speech-text pave connect'], ['Stefano Goria, Roseline Polle, Salvatore Fara, Nicholas Cummins', 'Revealing Confounding Biases: A Novel Benchmarking Approach for Aggregate-Level Performance Metrics in Health Assessments', 'goria24_interspeech', 'model assessment exaggerate overestimation overestimated machine alzheimer report cross-sectional small-scale'], ['Lucas Block Medin, Thomas Pellegrini, Lucile Gelin', "Self-Supervised Models for Phoneme Recognition: Applications in Children's Speech for Reading Learning", 'blockmedin24_interspeech', 'wavlm child base transformer non-english hubert various continue ctc wav'], ['Jules Cauzinille, Benoît Favre, Ricard Marxer, Dena Clink, Abdul Hamid Ahmad, Arnaud Rey', "Investigating self-supervised speech models' ability to classify animal vocalizations: The case of gibbon's vocal signatures", 'cauzinille24_interspeech', 'ssl probing pre-trained grey bioacoustic primate bird non-human classifier explainability'], ['Maurice Gerczuk, Shahin Amiriparian, Justina Lutz, Wolfgang Strube, Irina Papazova, Alkomiet Hasan, Björn W. Schuller', 'Exploring Gender-Specific Speech Patterns in Automatic Suicide Risk Assessment', 'gerczuk24_interspeech', 'patient modelling female gender-based psychiatric medicine specialised subject emergency hindered'], ['Thomas Rolland, Alberto Abad', 'Introduction To Partial Fine-tuning: A Comprehensive Evaluation Of End-to-end Children’s Automatic Speech Recognition Adaptation', 'rolland24_interspeech', 'asr dealing data departing granular overfit limited model challenge feedforward'], ['Thomas Rolland, Alberto Abad', 'Shared-Adapters: A Novel Transformer-based Parameter Efficient Transfer Learning Approach For Children’s Automatic Speech Recognition', 'rolland24b_interspeech', 'peft parameter-efficient asr fine-tuning pre-trained strike minimising model exceptional challenge'], ['Kentaro Seki, Shinnosuke Takamichi, Norihiro Takamune, Yuki Saito, Kanami Imamura, Hiroshi Saruwatari', 'Spatial Voice Conversion: Voice Conversion Preserving Spatial Information and Non-target Signals', 'seki24_interspeech', 'inherent waveform organize balancing bs stereo ignoring mixing encourage exploration'], ['Nina R. Benway, Jonathan L. Preston, Carol Espy-Wilson', 'Examining Vocal Tract Coordination in Childhood Apraxia of Speech with Acoustic-to-Articulatory Speech Inversion Feature Sets', 'benway24_interspeech', 'disorder auroc genetically neurodevelopmental correlation-based spatiotemporal nested sound -fold replicated'], ['Anna Stein, Kevin Tang', 'Modeling probabilistic reduction across domains with Naive Discriminative Learning', 'stein24_interspeech', 'word context syllable predictability competition cue outcome segment duration local'], ['Alkis Koudounas, Gabriele Ciravegna, Marco Fantini, Erika Crosetti, Giovanni Succo, Tania Cerquitelli, Elena Baralis', 'Voice Disorder Analysis: a Transformer-based Approach', 'koudounas24_interspeech', 'shortage pathology diagnosis data type solution under-explored recording private non-invasive'], ['KiHyun Nam, Hee-Soo Heo, Jee-weon Jung, Joonson Chung', 'Disentangled Representation Learning for Environment-agnostic Speaker Recognition', 'nam24b_interspeech', 'embedding auto-encoder framework extractor code popularly versatility utilises compatibility disentanglement'], ['Tomoki Koriyama', 'VAE-based Phoneme Alignment Using Gradient Annealing and SSL Acoustic Features', 'koriyama24_interspeech', 'vae model -based acoustic-feature mfa ctc-based boundary state-level linguistic widely-used'], ['Iva Ewert, Marvin Borsdorf, Haizhou Li, Tanja Schultz', 'Does the Lombard Effect Matter in Speech Separation? Introducing the Lombard-GRID-2mix Dataset', 'ewert24_interspeech', 'work normal reflexive soundscapes studied style speaking change two-speaker 
11146cocktail'], ['Cliodhna Hughes, Guy Brown, Ning Ma, Nicola Dibben', 'Acoustic Effects of Facial Feminisation Surgery on Speech and Singing: A Case Study', 'hughes24_interspeech', 'altered vocal tract people single-case procedure undergoing understudied undergo characteristic'], ['Gaëlle Laperrière, Sahar Ghannay, Bassam Jabaian, Yannick Estève', 'A dual task learning approach to fine-tune a multilingual semantic speech encoder for Spoken Language Understanding', 'laperriere24_interspeech', 'samu-xlsr slu enrichment specializing loss specialization language-agnostic vastly enrich portability'], ['Jacob Kealey, John R. Hershey, François Grondin', 'Unsupervised Improved MVDR Beamforming for Sound Enhancement', 'kealey24_interspeech', 'multi-channel separation data supervised isolated channel in-the-wild single distortionless case'], ['Yoshiaki Bando, Tomohiko Nakamura, Shinji Watanabe', 'Neural Blind Source Separation and Diarization for Distant Speech Recognition', 'bando24_interspeech', 'gss dsr multichannel supervision mixture method jointly separate weakly-supervised signal-level'], ['Jian Cheng', 'Context-Aware Speech Recognition Using Prompts for Language Learners', 'cheng24b_interspeech', 'gemini whisper spoken asr text context-awareness elicitors wer apps response'], ['Khalid Daoudi, Solange Milhé de Saint Victor, Alexandra Foubert-Samier, Margherita Fabbri, Anne Pavy-Le Traon, Olivier Rascol, Virginie Woisard, Wassilios G. Meissner', "Electroglottography for the assessment of dysphonia in Parkinson's disease and multiple system atrophy", 'daoudi24_interspeech', 'egg early msa-p parkinsonian msa diagnosis disorder stage analysis noninvasive'], ['Arunav Arya, Murtiza Ali, Karan Nathwani', 'Exploiting Wavelet Scattering Transform for an Unsupervised Speaker Diarization in Deep Neural Network Framework', 'arya24_interspeech', 'wst embeddings model speechbrain manner voxconverse segment-wise pyannote customize api'], ['Alexander Barnhill, Elmar Noeth, Andreas Maier, Christian Bergler', 'ANIMAL-CLEAN – A Deep Denoising Toolkit for Animal-Independent Signal Enhancement', 'barnhill24_interspeech', 'bioacoustic animal downstream largely bioacoustics myriad impede non-human existing recording'], ['Adriana Fernandez-Lopez, Honglie Chen, Pingchuan Ma, Lu Yin, Qiao Xiao, Stavros Petridis, Shiwei Liu, Maja Pantic', 'MSRS: Training Multimodal Speech Recognition Models from Scratch with Sparse Mask Optimization', 'fernandezlopez24_interspeech', 'dense vsr avsr regularization gradient transitioning stabilizes abbreviated vanishing non-zero'], ['James Tanner, Morgan Sonderegger, Jane Stuart-Smith, Tyler Kendall, Jeff Mielke, Robin Dodsworth, Erik Thomas', 'Exploring the anatomy of articulation rate in spontaneous English speech: relationships between utterance length effects and social factors', 'tanner24_interspeech', 'conditioned across gender age effect modulate leaving less broader shown'], ['Lila Kim, Cédric Gendrot', 'Using wav2vec 2.0 for phonetic classification tasks: methodological aspects', 'kim24l_interspeech', 'sequence phoneme longer correlating react speaker vector nasality airflow re
11146covering'], ['Zhaoyu Wang, Haohe Liu, Harry Coppock, Björn Schuller, Mark D. Plumbley', 'Neural Compression Augmentation for Contrastive Audio Representation Learning', 'wang24u_interspeech', 'nca self-supervised music underperform twin lossy audioset pivotal surpass generalisation'], ['Kwanghee Choi, Ankita Pasad, Tomohiko Nakamura, Satoru Fukayama, Karen Livescu, Shinji Watanabe', 'Self-Supervised Speech Representations are More Phonetic than Semantic', 'choi24b_interspeech', 'similarity pair datasets synonym curate corroborates snip property word linguistic'], ['Wiebke Hutiri, Tanvina Patel, Aaron Yi Ding, Odette Scharenborg', 'As Biased as You Measure: Methodological Pitfalls of Bias Evaluations in Speaker Verification Research', 'hutiri24_interspeech', 'base metric choice ratio-based hindering recommend contradictory group favour across'], ['Charles McGhee, Kate Knill, Mark Gales', 'Highly Intelligible Speaker-Independent Articulatory Synthesis', 'mcghee24_interspeech', 'synthesiser inversion deep real simulation-based wer synthetic training acoustic-to-articulatory struggle'], ['Xizi Wei, Stephen McGregor', 'Prompt Tuning for Speech Recognition on Unknown Spoken Name Entities', 'wei24_interspeech', 'entity phrase scenario reflecting named model unheard stratification pertains bot'], ['Rémi Uro, Marie Tahon, David Doukhan, Antoine Laurent, Albert Rilliard', 'Detecting the terminality of speech-turn boundary for spoken interactions in French TV and Radio content', 'uro24_interspeech', 'turn-taking terminal analyzing turn fusion place interrupting non-terminal floor initialization'], ['Liwei Liu, Huihui Wei, Dongya Liu, Zhonghua Fu', 'HarmoNet: Partial DeepFake Detection Network based on Multi-scale HarmoF0 Feature Fusion', 'liu24g_interspeech', 'add region audio track loss post-processor fake locating framework spoofing'], ['Nhan Phan, Anna von Zansen, Maria Kautonen, Ekaterina Voskoboinik, Tamas Grosz, Raili Hilden, Mikko Kurimo', 'Automated content assessment and feedback for Finnish L2 learners in a picture description speaking task', 'phan24_interspeech', 'grading asa language explanation low-resource automatic generation solution visual nlg'], ['Mathilde Hutin, Junfei Hu, Liesbeth Degand', 'Uh, um and mh: Are filled pauses prone to conversational converge?', 'hutin24_interspeech', 'speaker-oriented interlocutor frequent yet conversation asymmetrical interaction form participate debate'], ['Iona Gessinger, Bistra Andreeva, Benjamin R. Cowan', 'The Use of Modifiers and f0 in Remote Referential Communication with Human and Computer Partners', 'gessinger24_interspeech', 'competitor ground partner status condition description privileged common information responded'], ['Marvin Borsdorf, Zexu Pan, Haizhou Li, Tanja Schultz', 'wTIMIT2mix: A 
11146Cocktail Party Mixtures Database to Study Target Speaker Extraction for Normal and Whispered Speech', 'borsdorf24_interspeech', 'tse mode signal reference work two-speaker closed-set given equipped smart'], ['Louis Bahrman, Mathieu Fontaine, Jonathan Le Roux, Gaël Richard', 'Speech dereverberation constrained on room impulse response characteristics', 'bahrman24_interspeech', 'dereverberated rir regularizing dry signal loss acoustic black-box provision physically'], ['Xiang Li, Vivek Govindan, Rohit Paturi, Sundararajan Srinivasan', 'Speakers Unembedded: Embedding-free Approach to Long-form Neural Diarization', 'li24x_interspeech', 'eend embeddings speaker framework local embedding-based -pass additional mitigates generalizing'], ['Neelesh Samptur, Tanuka Bhattacharjee, Anirudh Chakravarty K, Seena Vengalil, Yamini Belur, Atchayaram Nalini, Prasanta Kumar Ghosh', 'Exploring Syllable Discriminability during Diadochokinetic Task with Increasing Dysarthria Severity for Patients with Amyotrophic Lateral Sclerosis', 'samptur24_interspeech', 'ddk al manual classification decline among automatic healthy cue impacted'], ['Benjamin Elie, David Doukhan, Rémi Uro, Lucas Ondel-Yang, Albert Rilliard, Simon Devauchelle', 'Articulatory Configurations across Genders and Periods in French Radio and TV archives', 'elie24b_interspeech', 'maeda frame gender female parameter protrusion diachronic assertion lowered spanning'], ['Lingyun Gao, Cristian Tejedor-Garcia, Helmer Strik, Catia Cucchiarini', 'Reading Miscue Detection in Primary School through Automatic Speech Recognition', 'gao24d_interspeech', 'whisper sota child exercise wav diagnosis vec asr highest dutch'], ['Michaela Watkins, Paul Boersma, Silke Hamann', 'Revisiting Pitch Jumps: F0 Ratio in Seoul Korean', 'watkins24_interspeech', 'octave periodicity vibration algorithm capture vocal-fold fortis downward upward chart'], ['Xuanjun Chen, Haibin Wu, Roger Jang, Hung-yi Lee', 'Singing Voice Graph Modeling for SingFake Detection', 'chen24o_interspeech', 'singer unseen sota model music groundbreaking mert struggled copyright deepfakes'], ['Chin-Yun Yu, György Fazekas', 'Differentiable Time-Varying Linear Prediction in the Context of End-to-End Analysis-by-Synthesis', 'yu24b_interspeech', 'generalise frame-wise vocoder efficient reconstructs time-invariant acceleration barrier source-filter recursive'], ['Xuanjun Chen, Jiawei Du, Haibin Wu, Jyh-Shing Roger Jang, Hung-yi Lee', 'Neural Codec-based Adversarial Sample Detection for Speaker Verification', 'chen24p_interspeech', 'asv sota codecs codec method single-model delivering surpassing re-synthesized defense'], ['Trung Dang, David Aponte, Dung Tran, Kazuhito Koishida', 'LiveSpeech: Low-Latency Zero-shot Text-to-Speech via Autoregressive Modeling of Audio Discrete Codes', 'dang24_interspeech', 'streaming codebook token codebooks codec grouping model-based hard enabling generative'], ['Adaeze Adigwe, Sarenne Wallbridge, Simon King', 'What do people hear? Listeners’ Perception of Conversational Speech', 'adigwe24_interspeech', 'tt explanation preference style underscoring pinpoint prompting synthesise organisation inappropriate'], ['Yifan Peng, Jinchuan Tian, William Chen, Siddhant Arora, Brian Yan, Yui Sudo, Muhammad Shakeel, Kwanghee Choi, Jiatong Shi, Xuankai Chang, Jee-weon Jung, Shinji Watanabe', 'OWSM v3.1: Better and Faster Open Whisper-Style Speech Models based on E-Branchformer', 'peng24b_interspeech', 'openai predecessor emergent toolkits reproducing biasing license data inferior zero-shot'], ['Kazutoshi Shinoda, Nobukatsu Hojo, Saki Mizuno, Keita Suzuki, Satoshi Kobashikawa, Ryo Masumura', 'Learning from Multiple Annotator Biased Labels in Multimodal Conversation', 'shinoda24_interspeech', 'bias minority class majority debiasing label overlook mitigates distribution value'], ['Shabnam Ghaffarzadegan, Luca Bondi, Wei-Chang Lin, Abinaya Kumar, Ho-Hsiang Wu, Hans-Georg Horst, Samarjit Das', 'Sound of Traffic: A Dataset for Acoustic Traffic Identification and Counting', 'ghaffarzadegan24_interspeech', 'vehicle zenodo org record http right-to-left radar passenger coil inductive'], ['Tobias Weise, Philipp Klumpp, Kubilay Can Demir, Paula Andrea Pérez-Toro, Maria Schuster, Elmar Noeth, Bjoern He
11146ismann, Andreas Maier, Seung Hee Yang', 'Speaker- and Text-Independent Estimation of Articulatory Movements and Phoneme Alignments from Speech', 'weise24_interspeech', 'aai inversion two-staged phoneme-related pta alignment frame aligner acoustic-to-articulatory task'], ['Zakaria Aldeneh, Takuya Higuchi, Jee-weon Jung, Skyler Seto, Tatiana Likhomanenko, Stephen Shum, Ahmed Hussen Abdelaziz, Shinji Watanabe, Barry-John Theobald', 'Can you Remove the Downstream Model for Speaker Recognition with Self-Supervised Speech Features?', 'aldeneh24_interspeech', 'verification simplify ingest filter-banks superb revisit sacrificing filter-bank performance inherently'], ['Donna Erickson, Albert Rilliard, Malin Svensson Lundmark, Adelaide Silva, Leticia Rebollo Couto, Oliver Niebuhr, João Antonio de Moraes', 'Collecting Mandible Movement in Brazilian Portuguese', 'erickson24_interspeech', 'closing quick speaker post-stress helmet prosodic structure sentence synchronization lowering'], ['Saurav Pahuja, Gabriel Ivucic, Pascal Himmelmann, Siqi Cai, Tanja Schultz, Haizhou Li', 'Leveraging Graphic and Convolutional Neural Networks for Auditory Attention Detection with EEG', 'pahuja24_interspeech', 'spatiotemporal selective ensemble spatial st-gcn single-trial prediction electroencephalography locus -second'], ['Seong-Gyun Leem, Daniel Fulford, Jukka-Pekka Onnela, David Gard, Carlos Busso', 'Keep, Delete, or Substitute: Frame Selection Strategy for Noise-Robust Speech Emotion Recognition', 'leem24_interspeech', 'ser noise framework background noisy dropped discarding emotionally suppressing avoiding'], ['Alkis Koudounas, Flavio Giobergia, Eliana Pastor, Elena Baralis', 'A Contrastive Learning Approach to Mitigate Bias in Speech Models', 'koudounas24b_interspeech', 'subgroup internal affected three-level overlooking user-defined fair imbalance adoption representation'], ['Debasmita Bhattacharya, Eleanor Lin, Run Chen, Julia Hirschberg', 'Switching Tongues, Sharing Hearts: Identifying the Relationship between Empathy and Code-switching in Speech', 'bhattacharya24_interspeech', 'csw multilingual motivation empathetic prior prevalence sociolinguistic qualitatively acoustic-prosodic asking'], ['Lucas Goncalves, Donita Robinson, Elizabeth Richerson, Carlos Busso', 'Bridging Emotions Across Languages: Low Rank Adaptation for Multilingual Speech Emotion Recognition', 'goncalves24_interspeech', 'ser lora expression surge envision taiwanese linguistic refining overcoming constantly'], ['Abinay Reddy Naini, Lucas Goncalves, Mary A. Kohler, Donita Robinson, Elizabeth Richerson, Carlos Busso', 'WHiSER: White House Tapes Speech Emotion Recognition Corpus', 'naini24_interspeech', 'ser speech-emotion complex authenticity defense healthcare advancing office offering perfect'], ['Joseph Liu, Mahesh Kumar Nandwana, Janne Pylkkönen, Hannes Heikinheimo, Morgan McGuire', 'Enhancing Multilingual Voice Toxicity Detection with Speech-Text Alignment', 'liu24h_interspeech', 'semantic classification classifier across general-purpose text framework ablation cross-modal conducting'], ['Sri Harsha Dumpala, Dushyant Sharma, Chandramouli Shama Sastry, Stanislav Kruchinin, James Fosburgh, Patrick A. Naylor', 'XANE: eXplainable Acoustic Neural Embeddings', 'dumpala24_interspeech', 'background detection parameter signal non-intrusive wavlm noise overlapped type method'], ['Spyretta Leivaditi, Tatsunari Matsushima, Matt Coler, Shekhar Nayak, Vass Verkhodanova', 'Fine-Tuning Strategies for Dutch Dysarthric Speech Recognition: Evaluating the Impact of Healthy, Disease-Specific, and Speaker-Specific Data', 'leivaditi24_interspeech', 'ssl third ineffective asr inadequate learning second widespread scarcity sufficiently'], ['Tanel Pärnamaa, Ando Saabas', 'Personalized Speech Enhancement Without a Separate Speaker Embedding Model', 'parnamaa24_interspeech', 'pse suppression winner teleconferencing icassp audio echo surpasses cancellation representation'], ['Deepanshu Gupta, Javier Latorre', 'Positional Description for Numerical Normalization ', 'gupta24d_interspeech', 'pd arithmetic digit model fatal mitigates tokenization tractable fst challenge'], ['Wangyou Zhang, Robin Scheibler, Kohei Saijo, Samuele Cornell, Chenda Li, Zhaoheng Ni, Jan Pirklbauer, Marvin Sach, Shinji Watanabe, Tim Fingscheidt, Yanmin Qian', 'URGENT Challenge: Universality, Robustness, and Generalizability For Speech Enhancement', 'zhang24h_interspeech', 'sub-tasks metric curated existing unify witnessed dereverberation data fill learning-based'], ['Kohei Saijo, Gordon Wichern, François G. Germain, Zexu Pan, Jonathan Le Roux', 'Enhanced Reverberation as Supervision for Unsupervised Speech Separation', 'saijo24_interspeech', 'ra mapped channel mixture era separated stable source-channel pseudo-targets loss'], ['ChengHung Hu, Yusuke Yasuda, Tomoki Toda', 'Embedding Learning for Preference-based Speech Quality Assessment', 'hu24d_interspeech', 'mo embeddings loss utterance closer similar preference score t-sne out-domain'], ['Pravin Mote, Berrak Sisman, Carlos Busso', 'Unsupervised Domain Adaptation for Speech Emotion Recognition using K-Nearest Neighbors Voice Conversion', 'mote24_interspeech', 'unlabeled labeled data ser transformed knn sample re-training model ineffective'], ['Catarina Botelho, John Mendonça, Anna Pompili, Tanja Schultz, Alberto Abad, Isabel Trancoso', "Macro-descriptors for Alzheimer's disease detection using large language models", 'botelho24_interspeech', 'llm transcription designated surpassing prompting experiment interpretable coherence annotator high-level'], ['Ali N. Salman, Zongyang Du, Shreeram Suresh Chandra, İsmail Rasim Ülgen, Carlos Busso, Berrak Sisman', 'Towards Naturalistic Voice Conversion: NaturalVoices Dataset with an Automatic Processing Pipeline', 'salman24_interspeech', 'natural speech expressive providing spontaneity msp-podcast sourced podcast podcasts scripted'], ['Muhammad Shakeel, Yui Sudo, Yifan Peng, Shinji Watanabe', 'Contextualized End-to-end Automatic Speech Recognition with Intermediate Biasing Loss', 'shakeel24_interspeech', 'contextual employing non-contextual contextualization layer unbiased objective ignore biased baseline'], ['Ailin Liu, Pepijn Vunderink, Jose Vargas Quiros, Chirag Raman, Hayley Hung', 'How Private is Low-Frequency Speech Audio in the Wild? An Analysis of Verbal Intelligibility by Humans and Machines', 'liu24i_interspeech', 'privacy social behavior delicate privacy-preserving wearable comprehensively ensures simulating balance'], ['Ladislav Mošner, Romain Serizel, Lukáš Burget, Oldřich Plchot, Emmanuel Vincent, Junyi Peng, Jan Černocký', 'Multi-Channel Extension of Pre-trained Models for Speaker Verification', 'mosner24_interspeech', 'ssl designing best-published enhancement interleaf downside noteworthy propagated cross-channel focus'], ['William Ravenscroft, George Close, Stefan Goetze, Thomas Hain, Mohammad Soleymanpour, Anurag Chowdhur
11146y, Mark C. Fuhs', 'Transcription-Free Fine-Tuning of Speech Separation Models for Noisy and Reverberant Multi-Speaker Automatic Speech Recognition', 'ravenscroft24_interspeech', 'asr pit signal-level loss reference training separator artefact often permutation'], ['Tsun-An Hsieh, Heeyoul Choi, Minje Kim', 'Multimodal Representation Loss Between Timed Text and Audio for Regularized Speech Separation', 'hsieh24b_interspeech', 'ttr summarizer semantics gap regularizer underexplored promotes wavlm text-based bert'], ['Wangyou Zhang, Kohei Saijo, Jee-weon Jung, Chenda Li, Shinji Watanabe, Yanmin Qian', 'Beyond Performance Plateaus: A Comprehensive Study on Scalability in Speech Enhancement', 'zhang24i_interspeech', 'architecture insight larger-scale small-sized size under-explored multi-domain budget plateau provide'], ['Jianyuan Sun, Wenwu Wang, Mark D. Plumbley', 'PFCA-Net: Pyramid Feature Fusion and Cross Content Attention Network for Automated Audio Captioning', 'sun24c_interspeech', 'aac scale fuse existing intricate across struggle facilitating high-dimensional top-down'], ['Woo-Jin Chung, Hong-Goo Kang', 'Speaker-Independent Acoustic-to-Articulatory Inversion through Multi-Channel Attention Discriminator', 'chung24_interspeech', 'aai model intricate overcoming kinematic pearson ema articulography attention-based electromagnetic'], ['Maria Teleki, Xiangjue Dong, Soohwan Kim, James Caverlee', 'Comparing ASR Systems in the Context of Speech Disfluencies', 'teleki24_interspeech', 'whisperx podcasts disfluent google non-scripted node podcast episode larger closer'], ['Ricardo García, Rodrigo Mahu, Nicolás Grágeda, Alejandro Luzanto, Nicolas Bohmer, Carlos Busso, Néstor Becerra Yoma', 'Speech emotion recognition with deep learning beamforming  on a distant human-robot interaction scenario', 'garcia24_interspeech', 'hri ser testing ccc average technology non-acted emulates indoor msp-podcast'], ['Sarthak Yadav, Zheng-Hua Tan', 'Audio Mamba: Selective State Spaces for Self-Supervised Audio Representations', 'yadav24_interspeech', 'general-purpose transformer spectrogram ssast spurred self-supervision size space model patch'], ['Brady Houston, Omid Sadjadi, Zejiang Hou, Srikanth Vishnubhotla, Kyu J. Han', 'Improving Multilingual ASR Robustness to Errors in Language Input', 'houston24_interspeech', 'sensitivity information inference model label common strategy susceptible continues ensuring'], ['Minh Nguyen, Franck Dernoncourt, Seunghyun Yoon, Hanieh Deilamsalehy, Hao Tan, Ryan Rossi, Quan Hung Tran, Trung Bui, Thien Huu Nguyen', 'Identifying Speakers in Dialogue Transcripts: A Text-based Approach Using Pretrained Language Models', 'nguyen24_interspeech', 'medium large-scale name speaker-identification encompassing speaker accessibility lacking novel archive'], ['Matthew Perez, Aneesha Sampath, Minxue Niu, Emily Mower Provost', 'Beyond Binary: Multiclass Paraphasia Detection with Generative Pretrained Transformers and End-to-End Models', 'perez24_interspeech', 'paraphasias gpt aphasia sequence automatic invention misuse single multiple focus'], ['Nicholas Klein, Tianxiang Chen, Hemlata Tak, Ricardo Casal, Elie Khoury', 'Source Tracing of Audio Deepfake Systems', 'klein24_interspeech', 'generation deepfakes anti-spoofing spoofing attribute discerning multi-language fake undergo genuine'], ['Christoph Boeddeker, Tobias Cord-Landwehr, Reinhold Haeb-Umbach', 'Once more Diarization: Improving meeting transcription systems through segment-level speaker reassignment', 'boeddeker24_interspeech', 'confusion enhancement correct revisiting error attribution assigning ease highlighting applicability'], ['Gabriel Pîrlogeanu, Octavian Pascu, Alexandru-Lucian Georgescu, Horia Cucu', 'Hybrid-Diarization System with Overlap Post-Processing for the DISPLACE 2024 Challenge', 'pirlogeanu24_interspeech', 'diarization der detection eval exhaustive participating collaborative overlapped semi-supervised ensemble'], ['Chris Bras, Tanvina Patel, Odette Scharenborg', 'Using articulated speech EEG signals for imagined speech decoding', 'bras24_interspeech', 'bcis amongst interface end transformative brain-computer point connecting electroencephalography vowel'], ['Emmy Phung, Harsh Deshpande, Ahmad Emami, Kanishk Singh', 'AR-NLU: A Framework for Enhancing Natural Language Understanding Model Robustness against ASR Errors', 'phung24_interspeech', 'nlu asr-robust gold transcript upstream pre-existing adversely challenge input intent'], ['Alireza Bayestehtashk, Amit Kumar, Mike Wurtz', 'Design of Feedback Active Noise Cancellation Filter Using Nested Recurrent Neural Networks', 'bayestehtashk24_interspeech', 'anc stability classical anti-noise problem iir stochastically intractable machine avenue'], ['Paige Tuttösí, H. Henny Yeung, Yue Wang, Fenqi Wang, Guillaume Denis, Jean-Julien Aucouturier, Angelica Lim', 'Mmm whatcha s
11146ay? Uncovering distal and proximal context effects in first and second-language word perception using psychophysical reverse correlation', 'tuttosi24_interspeech', 'surrounding profile pitch french timescales strikingly rate acoustic speaker vowel'], ['Alan Baade, Puyuan Peng, David Harwath', 'Neural Codec Language Models for Disentangled and Textless Voice Conversion', 'baade24_interspeech', 'speaker vall-e similarity any-to-any synthesis disentanglement normalizing less disentangle guidance'], ['Daniel Friedrichs, Monica Lancheros, Sam Kirkham, Lei He, Andrew Clark, Clemens Lutz, Volker Dellwo, Steven Moran', 'Temporal Co-Registration of Simultaneous Electromagnetic Articulography and Electroencephalography for Precise Articulatory and Neural Data Alignment', 'friedrichs24_interspeech', 'eeg ema planning precision onset synchronizes signal integrity kinematics event-related'], ['Octavian Pascu, Adriana Stan, Dan Oneata, Elisabeta Oneata, Horia Cucu', 'Towards generalisable and calibrated audio deepfake detection with self-supervised representations', 'pascu24_interspeech', 'generalisation reliable building desideratum rawnet model frozen less logistic coupled'], ['Suwon Shon, Kwangyoun Kim, Yi-Te Hsu, Prashant Sridhar, Shinji Watanabe, Karen Livescu', 'DiscreteSLU: A Large Language Model with Self-Supervised Discrete Speech Units for Spoken Language Understanding', 'shon24_interspeech', 'dsu llm encoder instruction-following adapter answering diverse task integration capability'], ['Wazeer Zulfikar, Nishat Protyasha, Camila Canales, Heli Patel, James Williamson, Laura Sarnie, Lisa Nowinski, Nataliya Kosmyna, Paige Townsend, Sophia Yuditskaya, Tanya Talkar, Utkarsh Oggy Sarawgi, Christopher McDougle, Thomas Quatieri, Pattie Maes, Maria Mody', 'Analyzing Speech Motor Movement using Surface Electromyography in Minimally Verbal Adults with Autism Spectrum Disorder', 'zulfikar24_interspeech', 'semg muscle skill facial greater gender-matched neurotypical discharge video-based correlation'], ['John Janiczek, Dading Chong, Dongyang Dai, Arlo Faria, Chao Wang, Tao Wang, Yuzong Liu', 'Multi-modal Adversarial Training for Zero-Shot Voice Cloning', 'janiczek24_interspeech', 'model magnified conditionally tt failing discriminates speech dataset fastspeech gan'], ['Xin Jing, Andreas Triantafyllopoulos, Björn Schuller', 'ParaCLAP – Towards a general language-audio model for computational paralinguistic tasks', 'jing24b_interspeech', 'clap audio query set pretraining relies surpass captioning available extending'], ['Michael Neumann, Hardik Kothare, Jackson Liscombe, Emma C.L. Leschly, Oliver Roesler, Vikram Ramanarayanan', 'Multimodal Digital Biomarkers for Longitudinal Tracking of Speech Impairment Severity in ALS: An Investigation of Clinically Important Differences', 'neumann24_interspeech', 'clinical remote disease meaningful assessment alsfrs-r responsiveness responsive change progression'], ['Sefik Emre Eskimez, Xiaofei Wang, Manthan Thakker, Chung-Hsien Tsai, Canrun Li, Zhen Xiao, Hemin Yang, Zirun Zhu, Min Tang, Jinyu Li, Sheng Zhao, Naoyuki Kanda', 'Total-Duration-Aware Duration Modeling for Text-to-Speech Systems', 'eskimez24_interspeech', 'tda adjusting diversity phoneme speech model total flow-matching quality intelligibility'], ['Keita Suzuki, Nobukatsu Hojo, Kazutoshi Shinoda, Saki Mizuno, Ryo Masumura', 'Participant-Pair-Wise Bottleneck Transformer for Engagement Estimation from Video Conversation', 'suzuki24_interspeech', 'multi-person token global participant multimodal stream attention among interaction sentiment'], ['Qiao Xiao, Pingchuan Ma, Adriana Fernandez-Lopez, Boqian Wu, Lu Yin, Stavros Petridis, Mykola Pechenizkiy, Maja Pantic, Decebal Constantin Mocanu, Shiwei Liu', 'Dynamic Data Pruning for Automatic Speech Recognition', 'xiao24b_interspeech', 'asr ever-growing barely prohibitively training speech-related entail save overhead granularity'], ['Aryan Chaudhary, Arshdeep Singh, Vinayak Abrol, Mark D. Plumbley', 'Efficient CNNs with Quaternion Transformations and Pruning for Audio Tagging', 'chaudhary24_interspeech', 'footprint memory cost algebra computational large-scale reduce resource-constrained audioset challenge'], ['Yatong Bai, Trung Dang, Dung Tran, Kazuhito Koishida, Somayeh Sojoudi', 'ConsistencyTTA: Accelerating Diffusion-Based Text-to-Audio Generation with Consistency Distillation', 'bai24b_interspeech', 'tta diffusion latent query inference classifier-free clap audiocaps cfg model'], ['Florian Lux, Sarina Meyer, Lyonel Behringer, Frank Zalkow, Phat Do, Matt Coler, Emanuël A. P. Habets, Ngoc Thang Vu', 'Meta Learning Text-to-Speech Synthesis in over 7000 Languages', 'lux24_interspeech', 'empower massively landscape releasing linguistic foster innovation zero-shot pretraining speech'], ['Xiaofei Wang, Sefik Emre Eskimez, Manthan Thakker, Hemin Yang, Zirun Zhu, Min Tang, Yufei Xia, Jinzhu Li, Sheng Zhao, Jinyu Li, Na
11146oyuki Kanda', 'An Investigation of Noise Robustness for Flow-Matching-Based Zero-Shot TTS', 'wang24v_interspeech', 'prompt audio pre-training quality generated strategy deteriorates mixing masked denoising'], ['Thomas Bott, Florian Lux, Ngoc Thang Vu', 'Controlling Emotion in Text-to-Speech with Natural Language Prompts', 'bott24_interspeech', 'prompt conditioned embeddings tractability steering emotionally prompting text merged intuitive'], ['Hao Yen, Pin-Jui Ku, Sabato Marco Siniscalchi, Chin-Hui Lee', 'Language-Universal Speech Attributes Modeling for Zero-Shot Multilingual Spoken Keyword Recognition', 'yen24_interspeech', 'skr phoneme-based seen character setting wer dat language sequence reduction'], ['Yanis Labrak, Adel Moumen, Richard Dufour, Mickael Rouvier', 'Zero-Shot End-To-End Spoken Question Answering In Medical Domain', 'labrak24_interspeech', 'sqa llm methodology resource transformative question-answering landscape accumulation resource-constrained underscore'], ['Jee-weon Jung, Wangyou Zhang, Jiatong Shi, Zakaria Aldeneh, Takuya Higuchi, Alex Gichamba, Barry-John Theobald, Ahmed Hussen Abdelaziz, Shinji Watanabe', 'ESPnet-SPK: full pipeline speaker embedding toolkit with reproducible recipes, self-supervised front-ends, and off-the-shelf models', 'jung24c_interspeech', 'embeddings extractor vox effortlessly effortless support use including simplifies facilitating'], ['Jing Pan, Jian Wu, Yashesh Gaur, Sunit Sivasankaran, Zhuo Chen, Shujie Liu, Jinyu Li', 'COSMIC: Data Efficient Instruction-tuning For Speech In-Context Learning', 'pan24b_interspeech', 'instruction-following -shot capability llm contextual sqa gpt- question-answer cost-effective biasing'], ['Weiran Wang, Zelin Wu, Diamantino Caseiro, Tsendsuren Munkhdalai, Khe Chai Sim, Pat Rondon, Golan Pundak, Gan Song, Rohit Prabhavalkar, Zhong Meng, Ding Zhao, Tara Sainath, Yanzhang He, Pedro Moreno Mengibar', 'Contextual Biasing with the Knuth-Morris-Pratt Matching Algorithm', 'wang24w_interspeech', 'bonus receives efficie
11146ncy vectorization search-based search trade cancel wfst gpu'], ['Tiantian Feng, Dimitrios Dimitriadis, Shrikanth S. Narayanan', 'Can Synthetic Audio From Generative Foundation Models Assist Audio Recognition and Speech Modeling?', 'feng24b_interspeech', 'speech-related usc-sail distance generation high-fidelity quality com heavily github assessing'], ['Grant Anderson, Emma Hart, Dimitra Gkatzia, Ian Beaver', 'Automated Human-Readable Label Generation in Open Intent Discovery', 'anderson24_interspeech', 'unlabelled extraction candidate cluster applying resorting dataset discovering analysing method'], ['Benjamin Barrera-Altuna, Daeun Lee, Zaima Zarnaz, Jinyoung Han, Seungbae Kim', 'The Interspeech 2024 TAUKADIAL Challenge: Multilingual Mild Cognitive Impairment Detection with Multimodal Approach', 'barreraaltuna24_interspeech', 'mci mortality speaking language dementia linguistic across worldwide scalability decline'], ['Ruchao Fan, Natarajan Balaji Shankar, Abeer Alwan', "Benchmarking Children's ASR with Supervised and Self-supervised Speech Foundation Models", 'fan24b_interspeech', 'finetuning sfms peft child wavlm whisper various benchmark state-ofthe-art stabilize'], ['Jee-weon Jung, Xin Wang, Nicholas Evans, Shinji Watanabe, Hye-jin Shim, Hemlata Tak, Siddhant Arora, Junichi Yamagishi, Joon Son Chung', 'To what extent can ASV systems naturally defend against spoofing attacks?', 'jung24d_interspeech', 'advancement spoofing-robust cutting-edge effortlessly necessitating acquires defense underscore non-target threat'], ['Ye Ni, Cong Pang, Chengwei Huang, Cairong Zou', 'MSA-DPCRN: A Multi-Scale Asymmetric Dual-Path Convolution Recurrent Network with Attentional Feature Fusion for Acoustic Echo Cancellation', 'ni24_interspeech', 'overlook deep-learning aec merge fuse numerous model validate majority maintaining'], ['Min Ma, Yuma Koizumi, Shigeki Karita, Heiga Zen, Jason Riesa, Haruko Ishikawa, Michiel Bacchiani', 'FLEURS-R: A Restored Multilingual Speech Corpus for Generation Tasks', 'ma24c_interspeech', 'fleurs restoration catalyze tt n-way few-shot language fidelity maintains quality'], ['Carly Demopoulos, Linnea Lampinen, Cristian Preciado, Hardik Kothare, Vikram Ramanarayanan', 'Preliminary Investigation of Psychometric Properties of a Novel Multimodal Dialog Based Affect Production Task in Children and Adolescents with Autism', 'demopoulos24_interspeech', 'apt age sex facial neurotypical ethnicity communication vocal race ability'], ['Shiyao Wang, Shiwan Zhao, Jiaming Zhou, Aobo Kong, Yong Qin', 'Enhancing Dysarthric Speech Recognition for Unseen Speakers via Prototype-Based Adaptation', 'wang24x_interspeech', 'dsr fine-tuning prototype per-word inconvenient formidable encapsulate disabled markedly hubert'], ['Shruti Palaskar, Ognjen Rudovic, Sameer Dharur, Florian Pesce, Gautam Krishna, Aswin Sivaraman, Jack Berkowitz, Ahmed Hussen Abdelaziz, Saurabh Adya, Ahmed Tewfik', 'Multimodal Large Language Models with Fusion Low Rank Adaptation for Device Directed Speech Detection', 'palaskar24_interspeech', 'llm fft eer pre-trained device-directed parity lower consume needing attains'], ['Lin Zhang, Xin Wang, Erica Cooper, Mireia Diez, Federico Landini, Nicholas Evans, Junichi Yamagishi', 'Spoof Diarization: &quot;What Spoofed When&quot; in Partially Spoofed Audio', 'zhang24j_interspeech', 'spoofing clustering partialspoof scenario pioneering task locating defines establishing localization'], ['Zehua Zhang, Xuyi Zhuang, Yukun Qian, Mingjiang Wang', 'Lightweight Dynamic Sparse Transformer for Monaural Speech Enhancement', 'zhang24k_interspeech', 'branch deep coarse fine block wb-pesq spectrum si-sdr feature aggregation'], ['Yu Tomita, Yingxiang Gao, Nobuaki Minematsu, Noriko Nakanishi, Daisuke Saito', 'Analysis and Visualization of Directional Diversity in Listening Fluency of World Englishes Speakers in the Framework of Mutual Shadowing', 'tomita24_interspeech', 'fluently listens disfluency passage communicability larger franca pronunciation shadowed lingua'], ['Yuxuan Xi, Yan Song, Lirong Dai, Haoyu Song, Ian McLoughlin', 'An Effective Local Prototypical Mapping Network for Speech Emotion Recognition', 'xi24_interspeech', 'prototype utterance-level frame-level optimized emotion-aware complex loss mer emotion-related backbone'], ['Yun Liu, Xuechen Liu, Xiaoxiao Miao, Junichi Yamagishi', 'Target Speaker Extraction with Curriculum Learning', 'liu24j_interspeech', 'tse selects interfer
11146ing increasing strategically libri similarity exceeded signal-to-distortion expose'], ['Jihyun Kim, Stijn Kindt, Nilesh Madhu, Hong-Goo Kang', 'Enhanced Deep Speech Separation in Clustered Ad Hoc Distributed Microphone Environments', 'kim24m_interspeech', 'ad-hoc transform-average-concatenate dual-path layer tailor blindly unpredictable learning accommodate fuse'], ['Behnam Gholami, Mostafa El-Khamy, KeeBong Song', 'Knowledge Distillation for Tiny Speech Enhancement with Latent Feature Augmentation', 'gholami24_interspeech', 'model teacher smaller student dnn voicebank complex resource-constrained deploying mimic'], ['Hardik Kothare, Michael Neumann, Cathy Zhang, Jackson Liscombe, Jordi W J van Unnik, Lianne C M Botman, Leonard H van den Berg, Ruben P A van Eijk, Vikram Ramanarayanan', 'How Consistent are Speech-Based Biomarkers in Remote Tracking of ALS Disease Progression Across Languages? A Case Study of English and Dutch', 'kothare24_interspeech', 'pal non-bulbar dutch-speaking bulbar onset english-speaking metric trajectory responsiveness amyotrophic'], ['Dongchao Yang, Dingdong Wang, Haohan Guo, Xueyuan Chen, Xixin Wu, Helen Meng', 'SimpleSpeech: Towards Simple and Efficient Text-to-Speech with Scalar Latent Transformer Diffusion Models', 'yang24l_interspeech', 'sq-codec nar speech-only space compact finite named speech tt non-'], ['Liu Xiaowang, Jinsong Zhang', 'A Study on the Information Mechanism of the 3rd Tone Sandhi Rule in Mandarin Disyllabic Words', 'xiaowang24_interspeech', 'lexical perspective sentence loss communicative minimal level definitive pair confusing'], ['Byeongjoo Ahn, Karren Yang, Brian Hamilton, Jonathan Sheaffer, Anurag Ranjan, Miguel Sarabia, Oncel Tuzel, Jen-Hao Rick Chang', 'Novel-view Acoustic Synthesis From 3D Reconstructed Rooms', 'ahn24c_interspeech', 'scene psnr dereverberation source sdr localization sound separation naively near-perfect'], ['Bence Mark Halpern, Thomas Tienkamp, Wen-Chin Huang, Lester Phillip Violeta, Teja Rebernik, Sebastiaan de Visscher, Max Witjes, Martijn Wieling, Defne Abur, Tomoki Toda', 'Quantifying the effect of speech pathology on automatic and human speaker verification', 'halpern24_interspeech', 'asv severity negatively perceptual performance correlated post-surgery surgical objective surgery'], ['Xinghao Huang, Weiwei Jiang, Long Rao, Wei Xu, Wenqing Cheng', 'Active Speaker Detection in Fisheye Meeting Scenes with Scene Spatial Spectrums', 'huang24g_interspeech', 'multi-party asd audio roundtable map on-screen circular dataset sota impressive'], ['Vishwanath Pratap Singh, Federico Malato, Ville Hautamäki, Md. Sahidullah, Tomi Kinnunen', 'ROAR: Reinforcing Original to Augmented Data Ratio Dynamics for Wav2vec2.0 Based ASR', 'singh24c_interspeech', 'augmentation heuristic librispeech dqn training amount balancing reinforcement recipe min'], ['Bin Zhao, Mingxuan Huang, Chenlu Ma, Jinyi Xue, Aijun Li, Kunyu Xu', 'Decoding Human Language Acquisition: EEG Evidence for Predictive Probabilistic Statistics in Word Segmentation', 'zhao24e_interspeech', 'stream tri-syllable glean auditory lobe lexical temporal statistical gyrus assertion'], ['Muhammad Yeza Baihaqi, Angel Garcia Contreras, Seiya Kawano, Koichiro Yoshino', 'Rapport-Driven Virtual Agent: Rapport Building Dialogue Strategy for Improving User Experience at First Meeting', 'baihaqi24_interspeech', 'free-form engagement satisfaction score human-agent naturalness correlation prompting collaborative llm'], ['Sahil Kumar, Jialu Li, Youshan Zhang', 'Vision Transformer Segmentation for Visual Bird Sound Denoising', 'kumar24_interspeech', 'vitvs contribution vit encompass persistent positioning long-range multi-scale audio struggle'], ['Vahid Khanagha, Dimitris Koutsaidis, Kaustubh Kalgaonkar, Sriram Srinivasan', 'Interference Aware Training Target for DNN based joint Acoustic Echo Cancellation and Noise Suppression', 'khanagha24_interspeech', 'double-talk shadowing near-end truth ground speech far-end aec spectral altering'], ['Yashish M. Siriwardena, Nathan Swedlow, Audrey Howard, Evan Gitterman, Dan Darcy, Carol Espy-Wilson, Andrea Fanelli', 'Accent Conversion with Articulatory Representations', 'siriwardena24_interspeech', 'speech non-native representation incorporate idea originates acoustic acoustic-to-articulatory phonetic used'], ['Amir Hussein, Desh Raj, Matthew Wiesner, Daniel Povey, Paola Garcia, Sanjeev Khudanpur', 'Enhancing Neural Transducer for Multilingual ASR with Synchronized Language Diarization', 'hussein24_interspeech', 'lid auxiliary synchronizes spanish-english seamless mandarin-english synchronize code-switching multitask switching'], ['Haolan Wang, Amin Edraki, Wai-Yip Chan, Iván López-Espejo, Jesper Jensen', 'No-Reference Speech Intelligibility Prediction Leveraging a Noisy-Speech ASR Pre-Trained Model', 'wang24y_interspeech', 'sip algorithm wav vec data-driven reference-based datasets parameter-efficient low-rank backbone'], ['Bao Hoang, Yijiang Pang, Hiroko Dodge, Jiayu Zhou', 'Translingual Language Markers for Cognitive Assessment from Spontaneous Speech', 'hoang24_interspeech', 'mci mmse detection treatment bilingual clinical taukadial prodromal enrichment mini-mental'], ['Hengchao Shang, Zongyao Li, Jiaxin Guo, Shaojun Li, Zhiqiang Rao, Yuanchang Luo, Daimeng Wei, Hao Yang', 'An End-to-End 
11146Speech Summarization Using Large Language Model', 'shang24_interspeech', 'ssum llm summary text connector abstractive long generate audio-text intricate'], ['Shuhua Li, Qirong Mao, Jiatong Shi', 'PL-TTS: A Generalizable Prompt-based Diffusion TTS Augmented by Large Language Model', 'li24y_interspeech', 'style diffusion-based libritts-r enhanced description grained control hot synthesis unsatisfactory'], ['Fabian Ritter-Gutierrez, Kuan-Po Huang, Jeremy H. M. Wong, Dianwen Ng, Hung-yi Lee, Nancy F. Chen, Eng-Siong Chng', 'Dataset-Distillation Generative Model for Speech Emotion Recognition', 'rittergutierrez24_interspeech', 'dataset iemocap downstream training reduces yet hinge accelerates size class'], ['Jiarui Hai, Karan Thakkar, Helin Wang, Zengyi Qin, Mounya Elhilali', 'DreamVoice: Text-Guided Voice Conversion', 'hai24_interspeech', 'timbre one-shot quot desired generation plugin libritts inclusive diffusion-based technology'], ['Yuwu Tang, Ziang Ma, Haitao Zhang', 'Enhanced Feature Learning with Normalized Knowledge Distillation for Audio Tagging', 'tang24b_interspeech', 'cnn-based lightweight transformer-based temperature model customized abundant backbone mainstream method'], ['George Joseph, Arun Baby', 'Speaker Personalization for Automatic Speech Recognition using Weight-Decomposed Low-Rank Adaptation', 'joseph24_interspeech', 'lora personalizing asr fine-tuning optimization model holy showcasing paramount limited'], ['Yinlin Guo, Yening Lv, Jinqiao Dou, Yan Zhang, Yuehai Wang', 'FLY-TTS: Fast, Lightweight and High-Quality End-to-End Text-to-Speech Synthesis', 'guo24c_interspeech', 'fourier model parameter-sharing convnext intel vits flow-based wavlm compress baseline'], ['Hanbin Bae, Pavel Andreev, Azat Saginbaev, Nicholas Babaev, WonJun Lee, Hosang Sung, Hoon-Young Cho', 'Speech Boosting: Low-Latency Live Speech Enhancement for TWS Earbuds', 'bae24_interspeech', 'on-device latency usage conversation complexity solution anc computational design wireless'], ['Tanya Talkar, Sherman Charles, Chelsea Krantsevich, Kan Kawabata', "Detection of Cognitive Impairment And Alzheimer's Disease Using a Speech- and Language-Based Protocol", 'talkar24_interspeech', 'pad mci auc tool speech-based individual administer presence medication blood'], ['Hyunjae Cho, Junhyeok Lee, Wonbin Jung', 'JenGAN: Stacked Shifted Filters in GAN-Based Speech Synthesis', 'cho24b_interspeech', 'artifact inference aliasing stacking non-autoregressive low-pass vocoders prevent audible evaluation'], ['Iwen E Kang, Christophe Van Gysel, Man-Hung Siu', 'Transformer-based Model for ASR N-Best Rescoring and Rewriting', 'kang24c_interspeech', 'rewrite rescore pertaining on-device privacy assistant exploring ensure increasingly engine'], ['Jaewon Kim, Won-Gook Choi, Seyun Ahn, Joon-Hyuk Chang', 'Sound of Vision: Audio Generation from Visual Text Embedding through Training Domain Discriminator', 'kim24n_interspeech', 'text-to-audio advancement tta aligns address inability adaptability fidelity compromise ensures'], ['Haojie Zhang, Tao Zhang, Ganjun Liu, Dehui Fu, Xiaohui Hou, Ying Lv', 'DysArinVox: DYSphonia &amp; DYSarthria mandARIN speech corpus', 'zhang24l_interspeech', 'chinese pathological comprehensive meticulously imagery diagnostics crafted diagnosed facilitating ensuring'], ['Noumida A, Rajeev Rajan', 'Multi-label Bird Species Classification from Field Recordings using Mel_Graph-GCN Framework', 'a24_interspeech', 'graph deep cnn convolutional gcn macro specaugment mel-spectrograms mel-spectrogram neural'], ['Sai Srujana Buddi, Satyam Kumar, Utkarsh Sarawgi, Vineet Garg, Shivesh Ranjan, Ognjen Rudovic, Ahmed Hussen Abdelaziz, Saurabh Adya', 'Comparative Analysis of Personalized Voice Activity Detection Systems: Assessing Real-World Effectiveness', 'buddi24_interspeech', 'pvad assess vad comprehensive various metric understanding paramount context-aware technology'], ['Ho-Young Choi, Won-Gook Choi, Joon-Hyuk Chang', 'Retrieval-Augmented Classifier Guidance for Audio Generation', 'choi24c_interspeech', 'sampling incurred low-quality noise-free diffusion dcase retrieved pretraining retrieve acquire'], ['Payal Mohapatra, Shamika Likhite, Subrata Biswas, Bashima Islam, Qi Zhu', 'Missingness-resilient Video-enhanced Multimodal Disfluency Detection', 'mohapatra24_interspeech', 'modality video missing unified fusion accommodates assured resilient curate available'], ['Qifei Li, Yingming Gao, Yuhua Wen, Cong Wang, Ya Li', 'Enhancing Modal Fusion by Alignment and Label Matching for Multimodal Emotion Recognition', 'li24z_interspeech', 'mem mer audio-video information emotional multitask guiding learning arising framework'], ['Zhengyang Chen, Xuechen Liu, Erica Cooper, Junichi Yamagishi, Yanmin Qian', 'Generating Speakers by Prompting Listener Impressions for Pre-trained Multi-Speaker Text-to-Speech Systems', 'chen24q_interspeech', 'prompt trait multispeaker tt pretrained flow-matching texttospeech method module tailor'], ['Junxu Wang, Zhihua Fang, Liang He', 'Self-Supervised Speaker Verification with Mini-Batch Prediction Correction', 'wang24z_interspeech', 'pseudo-labels noisy re-clustering method rectified exponential batch learning bound determines'], ['Zhong Meng, Zelin Wu, Rohit Prabhavalkar, Cal Peyser, Weiran Wang, Nanxin Chen, Tara N. Sainath, Bhuvana Ramabhadran', 'Text Injection for Neural Contextual Biasing', 'meng24d_interspeech', 'cti unpaired asr wer mwer injected phrase speech-like speech-text model'], ['Zihan Pan, Tianchi Liu, Hardik B. Sailor, Qiongqiong Wang', 'Attentive Merging of Hidden Embeddings from Pre-trained Speech Model for Anti-spoofing Detection', 'pan24c_interspeech', 'wavlm transformer hierarchical behavior layer uncertain large notably multi-layer ssl'], ['Kentaro Onda, Joonyong Park, Nobuaki Minematsu, Daisuke Saito', 'A Pilot Study of GSLM-based Simulation of Foreign Accentuation Only Using Native Speech Corpora', 'onda24_interspeech', 'gslm accent language spoken process inputting mentally listens reproduction unit'], ['Donghyun Seong, Joon-Hyuk Chang', 'H4C-TTS: Leveraging Multi-Modal Historical Context for Conversational Text-to-Speech', 'seong24_interspeech', 'tt conversation encoder situation appropriate modeling recent contextually natural distinguishing'], ['Daryush D. Mehta, Jarrad H. Van Stan, Hamzeh Ghasemzadeh, Robert E. Hillman', 'Comparing ambulatory voice measures during daily life with brief laboratory assessments in speakers with and without vocal hyperfunction', 'mehta24_interspeech', 'recording hyperfunctional accelerometer habitual assess use accounted tilt relate pressure'], ['Li Li, Shogo Seki', 'Improved Remixing Process for Domain Adaptation-Based Speech Enhancement by Mitigating Data Imbalance in Signal-to-Noise Ratio', 'li24aa_interspeech', 'remixed underrepresented snr balanced teacher remixit encompass re
11146corded model dataset'], ['Yuke Lin, Ming Cheng, Fulin Zhang, Yingying Gao, Shilei Zhang, Ming Li', 'VoxBlink2: A 100K+ Speaker Recognition Corpus and the Open-Set Speaker-Identification Benchmark', 'lin24j_interspeech', 'dataset gallery afterward single-model explore encompassing categorize wild probe concrete'], ['A F M Saif, Lisha Chen, Xiaodong Cui, Songtao Lu, Brian Kingsbury, Tianyi Chen', 'M2ASR: Multilingual Multi-task Automatic Speech Recognition via Multi-objective Optimization', 'saif24_interspeech', 'training conflict supervised formulates model objective across multiple task degrading'], ['Liangyu Nie, Sudarsana Reddy Kadiri, Ruchit Agrawal', 'MMSD-Net: Towards Multi-modal Stuttering Detection', 'nie24_interspeech', 'uni-modal speech stuttered impediment disruption context-aware irregular -score neural integral'], ['Mingyue Shi, Huali Zhou, Qinglin Meng, Nengheng Zheng', 'DBD-CI: Doubling the Band Density for Bilateral Cochlear Implants', 'shi24c_interspeech', 'electrode odd alternately speech-in-noise side even stimulating ripple ace dichotic'], ['Peidong Wang, Jian Xue, Jinyu Li, Junkun Chen, Aswin Shanmugam Subramanian', 'Soft Language Identification for Language-Agnostic Many-to-One End-to-End Speech Translation', 'wang24aa_interspeech', 'input model linear source accomplish initialized ensures network specified keeping'], ['Takaaki Saeki, Soumi Maiti, Shinnosuke Takamichi, Shinji Watanabe, Hiroshi Saruwatari', 'SpeechBERTScore: Reference-Aware Automatic Evaluation of Speech Generation Leveraging NLP Evaluation Metrics', 'saeki24_interspeech', 'subjective bertscore gold dense human computes cross-lingual applicability growing opinion'], ['Kang Zhu, Cunhang Fan, Jianhua Tao, Zhao Lv', 'Prompt Link Multimodal Fusion in Multimodal Sentiment Analysis', 'zhu24_interspeech', 'cpl linkage spl connecting modality dimension distance channel randomness connects'], ['Min-Han Shih, Ho-Lam Chung, Yu-Chi Pai, Ming-Hao Hsu, Guan-Ting Lin, Shang-Wen Li, Hung-yi Lee', 'GSQA: An End-to-End Model for Generative Spoken Question Answering', 'shih24b_interspeech', 'abstractive extractive sqa dataset answer empowers directly stride text surpasses'], ['Yiyang Zhao, Shuai Wang, Guangzhi Sun, Zehua Chen, Chao Zhang, Mingxing Xu, Thomas Fang Zheng', 'Whisper-PMFA: Partial Multi-Scale Feature Aggregation for Speaker Verification using Whisper Models', 'zhao24f_interspeech', 'block eer encoder pmfa cn-celeb low-rank eers ecapa-tdnn receiving resnet'], ['Anna Oura, Hideaki Kikuchi, Tetsunori Kobayashi', 'Preprocessing for acoustic-to-articulatory inversion using real-time MRI movies of Japanese speech', 'oura24_interspeech', 'rtmri aai normalization filtering articulatory estimation extraneous resemble indirect wavelet'], ['Yuting Yang, Guodong Ma, Yuke Li, Binbin Du, Haoqi Zhu, Liang Ruan', 'Learning from Back Chunks: Acquiring More Future Knowledge for Streaming ASR Models via Self Distillation', 'yang24m_interspeech', 'long-distance look-ahead aishell- later latency window information contextual fat chunk-based'], ['Yakun Song, Zhuo Chen, Xiaofei Wang, Ziyang Ma, Guanrou Yang, Xie Chen', 'TacoLM: GaTed Attention Equipped Codec Language Model are Efficient Zero-Shot Text to Speech Synthesizers', 'song24b_interspeech', 'inference efficiency speed vall-e layer cross-attention auto-regressive neural demo suffers'], ['Yaoxun Xu, Shi-Xiong Zhang, Jianwei Yu, Zhiyong Wu, Dong Yu', 'Comparing Discrete and Continuous Space LLMs for Speech Recognition', 'xu24d_interspeech', 'asr llama llm-based open-sourced organizing language advancing hubert achievement representation'], ['Chun Yin, Tai-Shih Chi, Yu Tsao, Hsin-Min Wang', 'SVSNet+: Enhancing Speaker Voice Similarity Assessment Models with Representations from Speech Foundation Models', 'yin24b_interspeech', 'wavlm sfms sfm pre-trained downstream incorporating performance dataset thoroughly baseline'], ['Oliver Roesler, Jackson Liscombe, Michael Neumann, Hardik Kothare, Abhishek Hosamath, Lakshmi Arbatti, Doug Habberstad, Christiane Suendermann-Oeft, Meredith Bartlett, Cathy Zhang, Nikhil Sukhdev, Kolja Wilms, Anusha Badathala, Sandrine Istas, Steve Ruhmel, Bryan Hansen, Madeline Hannan, David Henley, Arthur Wallace, Ira Shoulson, David Suendermann-Oeft, Vikram Ramanarayanan', 'Towards Scalable Remote Assessment of Mild Cognitive Impairment Via Multimodal Dialog', 'roesler24_interspeech', 'mci patient control administering liked reported biomarkers orofacial self-reported cloud-based'], ['Nahomi Kusunoki, Yosuke Higuchi, Tetsuji Ogawa, Tetsunori Kobayashi', 'Hierarchical Multi-Task Learning with CTC and Recursive Operation', 'kusunoki24_interspeech', 'hmtl intermediate layer prediction model lower-level subwords balancing asr target'], ['Eva Szekely, Maxwell Hope', 'An inclusive approach to creating a palette of synthetic voices for gender diversity', 'szekely24_interspeech', 'tt speaker identity expansive gender-independent failing vocal sgd emergent seeking'], ['Dail Kim, Da-Hee Yang, Donghyun Kim, Joon-Hyuk Chang, Jeonghwan Choi, Moa Lee, Jaemo Yang, Han-gil Moon', 'Guided conditioning with predictive network on score-based diffusion model for speech enhancement', 'kim24o_interspeech', 'removal trade-off diffusion-based guiding noise highlighted method emerged reflects outperforming'], ['Osamu Take, Shinnosuke Takamichi, Kentaro Seki, Yoshiaki Bando, Hiroshi Saruwatari', 'SaSLaW: Dialogue Speech Corpus with Audio-visual Egocentric Information Toward Environment-adaptive Dialogue Speech Synthesis', 'take24_interspeech', 'environment audio diverse spontaneous text-to-speech communication adaptation watch seamless model'], ['Hanzhao Li, Liumeng Xue, Haohan Guo, Xinfa Zhu, Yuanjun Lv, Lei Xie, Yunlin Chen, Hao Yin, Zhifei Li', 'Single-Codec: Single-Codebook Speech Codec towards High-Performance Speech Generation', 'li24ba_interspeech', 'multi-codebook module discrete encodec vq-vae time-invariant resampling decouple upsampling downsampling'], ['Woon-Haeng Heo, Joongyu Maeng, Yoseb Kang, Namhyun Cho', 'Centroid Estimation with Transformer-Based Speaker Embedder for Robust Target Speaker Extraction', 'heo24_interspeech', 'tse separator enrollment aiding information division separating stability utterance speech'], ['Xinyi Wu, Changqing Xu, Nan Li, Rongfeng Su, Lan Wang, Nan Yan', 'Depression Enhances Internal Inconsistency between Spoken and Semantic Emotion: Evidence from the Analysis of Emotion Expression in Conversation', 'wu24j_interspeech', 'depressed expressed modality healthy consistency talk neutral patient emotional topic'], ['Yan Xiong, Visar Berisha, Julie Liss, Chaital
11146i Chakrabarti', 'Improving Speech-Based Dysarthria Detection using Multi-task Learning with Gradient Projection', 'xiong24_interspeech', 'mtl clinical conflict size diagnostics single-task adversely task analytic task-specific'], ['Christina TÃ¥nnander, Shivam Mehta, Jonas Beskow, Jens Edlund', 'Beyond graphemes and phonemes: continuous phonological features in neural text-to-speech synthesis', 'tannander24_interspeech', 'confirming tt matcha-tts position perception change phoneme monotonic dual gradual'], ['Peter Wu, Ryan Kaveh, Raghav Nautiyal, Christine Zhang, Albert Guo, Anvitha Kachinthaya, Tavish Mishra, Bohan Yu, Alan W Black, Rikky Muller, Gopala Krishna Anumanchipalli', 'Towards EMG-to-Speech with Necklace Form Factor', 'wu24k_interspeech', 'emg electrode neck placed reveal device dry inconvenient electromyography regularly'], ['Tuan Nguyen, Huy Dat Tran', 'LingWav2Vec2: Linguistic-augmented wav2vec 2.0 for Vietnamese Mispronunciation Detection', 'nguyen24b_interspeech', 'canonical phoneme mha pronunciation top- feeding challenge linguistic multi-head adoption'], ['Jinyu Li, Leonardo Lancia', 'A multimodal approach to study the nature of coordinative patterns underlying speech rhythm', 'li24ca_interspeech', 'prominence language-specific production coordination sensorimotor speaker pre-recorded syllable defines unfamiliar'], ['Vahid Noroozi, Zhehuai Chen, Somshubra Majumdar, Steve Huang, Jagadeesh Balam, Boris Ginsburg', 'Instruction Data Generation and Unsupervised Adaptation for Speech Language Models', 'noroozi24_interspeech', 'synthetic generate sample text emerges component expand cross-modal large scarcity'], ['Shaoxiang Dang, Tetsuya Matsumoto, Yoshinori Takeuchi, Takashi Tsuboi, Yasuhiro Tanaka, Daisuke Nakatsubo, Satoshi Maesawa, Ryuta Saito, Masahisa Katsuno, Hiroaki Kudo', 'Developing vocal system impaired patient-aimed voice quality assessment approach using ASR representation-included multiple features', 'dang24b_interspeech', 'stn-dbs patient clinical predicting extraordinary showcasing hurdle immense deep pcc'], ['Xuyuan Li, Zengqiang Shang, Peiyang Shi, Hua Hua, Ta Li, Pengyuan Zhang', 'Expressive paragraph text-to-speech synthesis with multi-step variational autoencoder', 'li24da_interspeech', 'ep-mstts speech sliced variable vits-based style model audiobook challenge blizzard'], ['Shivam Mehta, Harm Lameris, Rajiv Punmiya, Jonas Beskow, Eva Szekely, Gustav Eje Henter', 'Should you use a probabilistic duration model in TTS? Probably! Especially for spontaneous speech', 'mehta24b_interspeech', 'nar modelling regression-based speaks non-autoregressive ignore deterministic sampled converting aloud'], ['Haitong Sun, Jaehyun Choi, Nobuaki Minematsu, Daisuke Saito', 'Acceleration of Posteriorgram-based DTW by Distilling the Class-to-class Distances Encoded in the Classifier Used to Calculate Posteriors', 'sun24d_interspeech', 'frame-to-frame posteriorgram n-dimensional distance bhattacharyya class distill fly often sequence'], ['Xu Li, Qirui Wang, Xiaoyu Liu', 'MaskSR: Masked Language Model for Full-band Speech Restoration', 'li24ea_interspeech', 'restoring token distortion reconstructs clipping sub-tasks extracted reverb target high'], ['Sathvik Udupa, Jesuraj Bandekar, Saurabh Kumar, Deekshitha G, Sandhya B, Abhayjeet S, Savitha Murthy, Priyanka Pai, Srinivasa Raghavan, Raoul Nanavati, Prasanta Kumar Ghosh', 'Adapter pre-training for improved speech recognition in unseen domains using low resource adapter tuning of self-supervised models', 'udupa24_interspeech', 'fine-tune ssl low-resource language decoding source target method large cer'], ['Daniel Galvez, Vladimir Bataev, Hainan Xu, Tim Kaldewey', 'Speed of Light Exact Greedy Decoding for RNN-T Speech Recognition Models on GPU', 'galvez24_interspeech', 'billion idle cuda quot implementation transducer inference end-to-end gpu-based parameter'], ['Melanie Weirich, Daniel Duran, Stefanie Jannedy', 'Gender and age based f0-variation in the German Plapper Corpus', 'weirich24_interspeech', 'femininity germany participant mean speaker difference effect masculinity donated female'], ['Han EunGi, Oh Hyun-Bin, Kim Sung-Bin, Corentin Nivelet Etcheberry, Suekyeong Nam, Janghoon Ju, Tae-Hyun Oh', 'Enhancing Speech-Driven 3D Facial Animation with Audio-Visual Guidance from Lip Reading Expert', 'eungi24_interspeech', 'loss motion movement garnered overlook generate perceptual realism readability cost-effective'], ['Liangwei Chen, Xiren Zhou, Qiang Tu, Huanhuan Chen', 'Enhancing Speech and Music Discrimination Through the Integration of Static and Dynamic Features', 'chen24r_interspeech', 'audio raw classification reservoir speech-music readout extracted evolve autoencoders stacked'], ['Yan Wan, Mengyi Sun, Xinchen Kang, Jingting Li, Pengfei Guo, Ming Gao, Su-Jing Wang', 'CDSD: Chinese Dysarthria Speech Database', 'w
11146an24b_interspeech', 'dysarthric cer formidable individual impacting html socially featuring dsr widespread'], ['Qingye Shen, Leonardo Lancia, Noel Nguyen', 'A novel experimental design for the study of listener-to-listener convergence in phoneme categorization', 'shen24c_interspeech', 'listener endpoint game constraint identify psychometric comply unambiguously sound avenue'], ['Yuchun Shu, Bo Hu, Yifeng He, Hao Shi, Longbiao Wang, Jianwu Dang', 'Error Correction by Paying Attention to Both Acoustic and Confidence References for Automatic Speech Recognition', 'shu24_interspeech', 'asr wrong hypothesis n-best well-founded word cross-attention non-autoregressive edit recovering'], ['Meiling Chen, Pengjie Liu, Heng Yang, Haofeng Wang', 'Towards End-to-End Unified Recognition for Mandarin and Cantonese', 'chen24s_interspeech', 'cer mandarin-only efficiency doubled coexist training system model high-resource demanding'], ['Ted Kye', 'Affricates in Lushootseed', 'kye24_interspeech', 'salish ejective coast voiced acoustic-articulatory affricate diachronic frication gravity typological'], ['Zeyang Song, Qianhui Liu, Qu Yang, Yizhou Peng, Haizhou Li', 'ED-sKWS: Early-Decision Spiking Neural Networks for Rapid, and Energy-Efficient Keyword Spotting', 'song24c_interspeech', 'kw energy consumption command speech efficiency snns snn end timestamp'], ['Zimeng Li, Zhongxuan Mao, Shengting Shen, Ivan Yuen, Ping Tang', 'The Production of Contrastive Focus by  7 to 13-year-olds Learning Mandarin Chinese', 'li24fa_interspeech', 'age revealed acquire discourse child tonal perceived tone lexical conveys'], ['Sarina Meyer, Florian Lux, Ngoc Thang Vu', 'Probing the Feasibility of Multilingual Speaker Anonymization', 'meyer24_interspeech', 'privacy language globe speech restricts anonymized protect deterioration component english'], ['Yuejiao Wang, Xianmin Gong, Lingwei Meng, Xixin Wu, Helen Meng', 'Large Language Model-based FMRI Encoding of Language Functions for Subjects with Neurocognitive Disorder', 'wang24ba_interspeech', 'ncd brain language-related functional score cognitive older correlation change adult'], ['Yue Gu, Zhihao Du, Shiliang Zhang, jiqing Han, Yongjun He', 'Personality-memory Gated Adaptation: An Efficient Speaker Adaptation for Personalized End-to-end Automatic Speech Recognition', 'gu24b_interspeech', 'adapter target generalization asr backbone personality cer encoder sacrifice vocal'], ['Daniel Haider, Felix Perfler, Vincent Lostanlen, Martin Ehler, Peter Balazs', 'Hold Me Tight: Stable Encoder-Decoder Design for Speech Enhancement', 'haider24_interspeech', 'audio encoder adapt guaranteeing conservation filter solution preprocess low-complexity objective'], ['Muhammad Umer Sheikh, Hassan Abid, Bhuiyan Sanjid Shafique, Asif Hanif, Muhammad Haris Khan', 'Bird Whisperer: Leveraging Large Pre-trained Acoustic Model for Bird Call Classification', 'sheikh24_interspeech', 'whisper adapting bioacoustic non-human categorizing feature underscore -score imbalance human'], ['Sylvain Coulange, Tsuneo Kato, Solange Rossato, Monica Masperi', 'Exploring Impact of Pausing and Lexical Stress Patterns on L2 English Comprehensibility in Real Time', 'coulange24_interspeech', 'pause click stressed incorrectly protocol real-time struggling sixty listener word'], ['Rhiannon Mogridge, Anton Ragni', 'Learning from memory-based models', 'mogridge24_interspeech', 'cpc psychology memory parametric field equivalently feature task eliminated sacrificing'], ['Chan-yeong Lim, Hyun-seo Shin, Ju-ho Kim, Jungwoo Heo, Kyo-Won K
11146oo, Seung-bin Kim, Ha-Jin Yu', 'Improving Noise Robustness in Self-supervised Pre-trained Model for Speaker Verification', 'lim24_interspeech', 'warm-up distorting loss noisy information teacher-student strategy angular unexplored prototypical'], ['Jiajun He, Tomoki Toda', '2DP-2MRC: 2-Dimensional Pointer-based Machine Reading Comprehension Method for Multimodal Moment Retrieval', 'he24_interspeech', 'clip-based coarse-grained video underperforms overlooking pointer imprecise existing heavy categorized'], ['Qiang Fang', 'On The Performance of EMA-synchronized Speech and Stand-alone Speech in Acoustic-to-articulatory Inversion', 'fang24_interspeech', 'aai synchronized degrade articulatory root-mean-square acoustic-articulatory pearson various ema latest'], ['Raj Gothi, Rahul Kumar, Mildred Pereira, Nagesh Nayak, Preeti Rao', 'A Dataset and Two-pass System for Reading Miscue Detection', 'gothi24_interspeech', 'asr wide-ranging resource-intensive diagnostics mispronounced elementary limiting alternate viewed apart'], ['Yuxin Xie, Zhihong Zhu, Xianwei Zhuang, Liming Liang, Zhichang Wang, Yuexian Zou', 'GPA: Global and Prototype Alignment for Audio-Text Retrieval', 'xie24c_interspeech', 'coarse-grained audio text atr pair fine-grained interaction similarity justify pursue'], ['Yaqian Hao, Chenguang Hu, Yingying Gao, Shilei Zhang, Junlan Feng', 'On Calibration of Speech Classification Models: Insights from Energy-Based Model Investigations', 'hao24_interspeech', 'overconfidence calibrating task guaranteeing classifier deep decision-making mitigating sacrificing emphasizes'], ['Taisei Omine, Kenta Akita, Reiji Tsuruno', 'Robust Laughter Segmentation with Automatic Diverse Data Synthesis', 'omine24_interspeech', 'audio arbitrary annotation annotates datasets training necessitates laugh automatically annotate'], ['Wing-Zin Leung, Mattias Cross, Anton Ragni, Stefan Goetze', 'Training Data Augmentation for Dysarthric Automatic Speech Recognition by Text-to-Dysarthric-Speech Synthesis', 'leung24_interspeech', 'dasr asr diffusion-based fine-tuning pwd augmentative aac limited finetuning dysarthria'], ['Denise Moussa, Sandra Bergmann, Christian Riess', 'Unmasking Neural Codecs: Forensic Identification of AI-compressed Speech', 'moussa24_interspeech', 'audio ai-based forensics compression lossy trace neurally towards encodec cross-dataset'], ['Yaqian Hao, Chenguang Hu, Yingying Gao, Shilei Zhang, Junlan Feng', 'Exploring Energy-Based Models for Out-of-Distribution Detection in Dialect Identification', 'hao24b_interspeech', 'ood energy score joint enhance conclusively loss sharpness model confronted'], ['Hao Li, Yuan Fang, Xueliang Zhang, Fei Chen, Guanglai Gao', 'Cross-Attention-Guided WaveNet for EEG-to-MEL Spectrogram Reconstruction', 'li24ga_interspeech', '-band mel eeg granularity modality enhance mixup correlation combined cross-attention'], ['Raul Monteiro', 'Adding User Feedback To Enhance CB-Whisper', 'monteiro24_interspeech', 'ov-kws biasing keywords tt keyword-spotting open-vocabulary classifier non-trivial assess audio'], ['Sarah Wesolek, Piotr Gulgowski, Joanna Blaszczak, Marzena Zygis', 'The influence of L2 accent strength and different error types on personality trait ratings', 'wesolek24_interspeech', 'warmth competence grammatical less phonological substitution towards impact unfavorable graded'], ['Ashishkumar Gudmalwar, Nirmesh Shah, Sai Akarsh, Pankaj Wasnik, Rajiv Ratn Shah', 'VECL-TTS: Voice identity and Emotional style controllable Cross-Lingual Text-to-Speech', 'gudmalwar24_interspeech', 'tt emotion dubbing marathi necessitates introduce limited language telugu transferring'], ['Jenifer Vega Rodriguez, Nathalie Vallée, Christophe Savariaux, Silvain Gerber', 'Nasal Air Flow During Speech Production In Korebaju', 'vegarodriguez24_interspeech', 'consonant oral colombia amazonian implosive eva ejective coe harmony carryover'], ['Juhwan Yoon, WooSeok Ko, Seyun Um, Sungwoong Hwang, Soojoong Hwang, Changhwan Kim, Hong-Goo Kang', 'UNIQUE : Unsupervised Network for Integrated Speech Quality Evaluation', 'yoon24b_interspeech', 'score metric manner synthetic anomaly objective measure comprising sophisticated paired'], ['Xuenan Xu, Haohe Liu, Mengyue Wu, Wenwu Wang, Mark D. Plumbley', 'Efficient Audio Captioning with Encoder-Level Knowledge Distillation', 'xu24e_interspeech', 'aac loss mse contrastive data-scarce sequence-level model distill exhibiting performance'], ['Zhiyong Wang, Ruibo Fu, Zhengqi Wen, Yuankun Xie, Yukun Liu, Xiaopeng Wang, Xuefei Liu, Yongwei Li, Jianhua Tao, Xin Qi, Yi Lu, Shuchen Shi', 'Generalized Fake Audio Detection via Deep Stable Learning', 'wang24ca_interspeech', 'swl extra datasets distribution training shift weight decorrelating complicate sample'], ['Haoxiang Shi, Ziqi Liang, Jun Yu', 'Emotional Cues Extraction and Fusion for Multi-modal Emotion Prediction and Recognition in Conversation', 'shi24d_interspeech', 'epc stage modality erc modality-specific overlooking meld forthcoming forecast learning'], ['Xuenan Xu, Pingyue Zhang, Ming Yan, Ji Zhang, Mengyue Wu', 'Enhancing Zero-shot Audio Classification using Sound Attribute Knowledge from Large Language Models', 'xu24f_interspeech', 'class description label innate audioset multi-dimensional learning relied ablation never'], ['Neha Sahipjohn, Ashishkumar Gudmalwar, Nirmesh Shah, Pankaj Wasnik, Rajiv Ratn Shah', 'DubWise: Video-Guided Speech Duration Control in Multimodal LLM-based Text-to-Speech for Dubbing', 'sahipjohn24_interspeech', 'lip token text different language video tt sync via cloning'], ['Zeyu Xie, Baihan Li, Xuenan Xu, Zheng Liang, Kai Yu, Mengyue Wu', 'FakeSound: Deepfake General Audio Detection', 'xie24d_interspeech', 'human discerning tester proliferation model dataset website surpasses locate viewed'], ['Tomi H. Kinnunen, Rosa Gonzalez Hautamäki, Xin Wang, Junichi Yamagishi', 'Speaker Detection by the Individual Listener and the Crowd: Parametric Models Applicable to Bonafide and Deepfake Speech', 'kinnunen24_interspeech', 'role-play subjective det eers fake set-up crowdsourcing spoofed less observing'], ['Ying Shi, Lantian Li, Shi Yin, Dong Wang, Jiqing Han', '
11146Serialized Output Training by Learned Dominance', 'shi24e_interspeech', '-mix pit multi-talker speech component showcased sot librimix time-based module'], ['Bolaji Yusuf, Jan Honza Cernocky, Murat Saraçlar', 'Pretraining End-to-End Keyword Search with Automatically Discovered Acoustic Units', 'yusuf24b_interspeech', 'aud untranscribed kw generally system complimentary asr-based finetuning data simplify'], ['Wei Liu, Jingyong Hou, Dong Yang, Muyong Cao, Tan Lee', 'LUPET: Incorporating Hierarchical Information Path into Multilingual ASR', 'liu24k_interspeech', 'lid layer routed design mitigates phoneme language high-resource high-performance linguistic'], ['Jing Xu, Minglin Wu, Xixin Wu, Helen Meng', 'Seamless Language Expansion: Enhancing Multilingual Mastery in Self-Supervised Models', 'xu24g_interspeech', 'existed ssl new preservation re-clustering impairing adaptation ability re-synthesis lora'], ['Martha Schubert, Daniel Duran, Ingo Siegert', 'Challenges of German Speech Recognition: A Study on Multi-ethnolectal Speech Among Adolescents', 'schubert24_interspeech', 'engine multi-ethnic openai meta nemo spontaneous nvidia underrepresented discern persist'], ['Taewoo Kim, Choonsang Cho, Young Han Lee', 'Period Singer: Integrating Periodic and Aperiodic Variational Autoencoders for Natural-Sounding End-to-End Singing Voice Synthesis', 'kim24p_interspeech', 'svs alignment corroborated one-to-many aligner address monotonic high-fidelity component owing'], ['Nao Hodoshima', 'Effects of talker and playback rate of reverberation-induced speech on speech intelligibility of older adults', 'hodoshima24_interspeech', 'oas intelligible reverberation fast original slow significantly twenty-four announcement headphone'], ['Swarup Ranjan Behera, Abhishek Dhiman, Karthik Gowda, Aalekhya Satya Narayani', 'FastAST: Accelerating Audio Spectrogram Transformer via Token Merging and Cross-Model Knowledge Distillation', 'behera24_interspeech', 'ast tome speed accuracy framework inference resource-efficient impact throughput compromising'], ['Zirui Ge, Xinzhou Xu, Haiyan Guo, Tingting Wang, Zhen Yang, Björn W. Schuller', 'DGPN: A Dual Graph Prototypical Network for Few-Shot Speech Spoofing Algorithm Recognition', 'ge24_interspeech', 'inter-speech impeding incipient showcasing ample depict realm anti-spoofing representation emerging'], ['Baihan Li, Zeyu Xie, Xuenan Xu, Yiwei Guo, Ming Yan, Ji Zhang, Kai Yu, Mengyue Wu', 'DiveSound: LLM-Assisted Automatic Taxonomy Construction for Diverse Audio Generation', 'li24ha_interspeech', 'diversity dataset multimodal framework subcategories text-to-audio visual overlook class diversified'], ['Zuzanna Miodonska, Michal Kręcichwost, Ewa Kwaśniok, Agata Sage, Pawel Badura', 'Frication noise features of Polish voiceless dental fricative and affricate produced by children with and without speech disorder', 'miodonska24_interspeech', 'interdental articulation band formant-related employed normative sibilant computer-aided disordered accompanying'], ['Ying Hu, Huamin Yang, Hao Huang, Liang He', 'Cross-modal Features Interaction-and-Aggregation Network with Self-consistency Training for Speech Emotion Recognition', 'hu24e_interspeech', 'ser multimodal modality supervise task-related shallower bimodal selective adaptively deeper'], ['Donghyun Seong, Hoyoung Lee, Joon-Hyuk Chang', 'TSP-TTS: Text-based Style Predictor with Residual Vector Quantization for Expressive Text-to-Speech', 'seong24b_interspeech', 'speech tt representation incorporating reference human-like quality fail conditioned regular'], ['Ziping Zhao, Tian Gao, Haishuai Wang, Björn Schuller', 'MFDR: Multiple-stage Fusion and Dynamically Refined Network for Multimodal Emotion Recognition', 'zhao24g_interspeech', 'perception window context strip cmu-mosei frame misalignment truncation emotionally discovering'], ['Bubai Maji, Rajlakshmi Guha, Aurobinda Routray, Shazia Nasreen, Debabrata Majumdar', 'Investigation of Layer-Wise Speech Representations in Self-Supervised Learning Models: A Cross-Lingual Study in Detecting Depression', 'maji24_interspeech', 'upstream wavlm hubert pooling add single-language mixed-language detection language max'], ['Yun Hao, Reihaneh Amooie, Wietse de Vries, Thomas Tienkamp, Rik van Noord, Martijn Wieling', 'Exploring Self-Supervised Speech Representations for Cross-lingual Acoustic-to-Articulatory Inversion', 'hao24c_interspeech', 'aai articulatory ssl data potential language prospect inferring less english'], ['Wei Liu, Jingyong Hou, Dong Yang, Muyong Cao, Tan Lee', 'A Parameter-efficient Language Extension Framework for Multilingual ASR', 'liu24l_interspeech', 'masr peft continual module candidate sub-problems add-on probabilistically low-resourced catastrophic'], ['Debasish Ray Mohapatra, Victor Zappi, Sidney Fels', '2.5D Vocal Tract Modeling: Bridging Low-Dimensional Efficiency with 3D Accuracy', 'mohapatra24b_interspeech', 'solver geometry symmetry model finite-difference blend mid-sagittal cross-sectional surpassing aligns'], ['Jesuraj Bandekar, Sathvik Udupa, Prasanta Kumar Ghosh', 'Articulatory synthesis using representations learnt through phonetic label-aware contrastive loss', 'bandekar24_interspeech', 'speech trajectory human-level framewise deep human-like learning learning-based baseline sequence-to-sequence'], ['Yongjie Si, Yanxiong Li, Jialong Li, Jiaxin Tan, Qianhua He', 'Fully Few-shot Class-incremental Audio Classification Using Expandable Dual-embedding Extractor', 'si24_interspeech', 'ast session prototype base fsc- embedding classifier training class sample'], ['Rastislav Rabatin, Frank Seide, Ernie Chang', 'Navigating the Minefield of MT Beam Search in Cascaded Streaming Speech Translation', 'rabatin24_interspeech', 'beam-search greedy handling intermediate real-time emitting final unequal machine anticipated'], ['Ling Dong, Zhengtao Yu, Wenjun Wang, Yuxin Huang, Shengxiang Gao, Guojiang Zhou', 'Integrating Speech Self-Supervised Learning Models and Large Language Models for ASR', 'dong24_interspeech', 'llm decoder-only garnered potential transport connecting mainstream aligning writing ssl'], ['Anbai Jiang, Bing Han, Zhiqiang Lv, Yufeng Deng, Wei-Qiang Zhang, Xie Chen, Yanmin Qian, Jia Liu, Pingyi Fan', 'AnoPatch: Towards Better Consistency in Machine Anomalous Sound Detection', 'jiang24c_interspeech', 'pre-trained asd audio dcase inconsistency vit fine-tunes datasets inductive model'], ['Nguyen Manh Tien Anh, Thach Ho Sy', 'Improving Speech Recognition with Prompt-based Contextualized ASR and LLM-based Re-predictor', 'manhtienanh24_interspeech', 'contextual mechanism bot system biasing adapter task text encounter task-specific'], ['Katelyn Taylor, Amelia Gully, Helena Daffern', 'Familiar and Unfamiliar Speaker Identification in Speech and Singing', 'taylor24_interspeech', 'social close network listener sung familiarity recognise sample foil listening'], ['Zihan Zhang, Xianjun Xia, Chuanzeng Huang, Yijian Xiao, Lei Xie', 'BS-PLCNet 2: Two-stage Band-split Packet Loss Concealment Network with Intra-model Knowledge Distillation', 'zhang24m_interspeech', 'plc icassp plcmos noncausal future complexity dual-path flop module computational'], ['Shuchen Shi, Ruibo Fu, Zhengqi Wen, Jianhua Tao, Tao Wang, Chunyu Qiang, Yi Lu, Xin Qi, Xuefei Liu, Yukun Liu, Yongwei Li, Zhiyong Wang, Xiaopeng Wang', 'PPPR: Portable Plug-in Prompt Refiner for Text to Audio Generation', 'shi24f_interspeech', 'tta description rich enhance text-to-audio inception surpassing altering playing faced'], ['Minglin Wu, Jing Xu, Xixin Wu, Helen Meng', 'Prompting Large Language Models with Mispronunciation Detection and Diagnosis Abilities', 'wu24l_interspeech', 'llm audio prompt english per encoder decoder phone mdd text'], ['Keigo Hojo, Yukoh Wakabayashi, Kengo Ohta, Atsunori Ogawa, Norihide Kitaoka', 'Boosting CTC-based ASR using inter-layer attention-based CTC loss', 'hojo24_interspeech', 'layer encoder intermediate output wer tedlium attention mechanism rtf dividing'], ['Dan Wells, Andrea Lorena Aldana Blanco, Cassia Valentini, Erica Cooper, Aidan Pine, Junichi Yamagishi, Korin Richmond', 'Experimental evaluation of MOS, AB and BWS listening test designs', 'wells24_interspeech', 'type listener seeming liked likeability counterbalanced re-use fastest questioned easiest'], ['Alon Vinnikov, Amir Ivry, Aviv Hurvitz, Igor Abramovski, Sharon K
11146oubi, Ilya Gurvich, Shai Peer, Xiong Xiao, Benjamin Martinez Elizalde, Naoyuki Kanda, Xiaofei Wang, Shalev Shaer, Stav Yagev, Yossi Asher, Sunit Sivasankaran, Yifan Gong, Min Tang, Huaming Wang, Eyal Krupka', 'NOTSOFAR-1 Challenge: New Datasets, Baseline, and Tasks for Distant Meeting Transcription', 'vinnikov24_interspeech', 'dasr benchmark launch attendee dataset office far-field -hour averaging multi-channel'], ['Jinpeng Li, Yu Pu, Qi Sun, Wei-Qiang Zhang', "Improving Whisper's Recognition Performance for Under-Represented Language Kazakh Leveraging Unpaired Speech and Text", 'li24ia_interspeech', 'pseudo-labeled data gpt researching hallucination worth low-cost penalty fine-tune ultimately'], ['Yunrui Cai, Zhiyong Wu, Jia Jia, Helen Meng', 'LoRA-MER: Low-Rank Adaptation of Pre-Trained Speech Models for Multimodal Emotion Recognition Using Mutual Information', 'cai24b_interspeech', 'mer challenge extract lora mine model feature frozen surpasses fine-tune'], ['Kim Sung-Bin, Lee Chae-Yeon, Gihun Son, Oh Hyun-Bin, Janghoon Ju, Suekyeong Nam, Tae-Hyun Oh', 'MultiTalk: Enhancing 3D Talking Head Generation Across Languages with Multilingual Video Dataset', 'sungbin24_interspeech', 'lip-sync movement datasets speech-driven convincing covering comprising github enhances mouth'], ['Wenjun Wang, Shangbin Mo, Ling Dong, Zhengtao Yu, Junjun Guo, Yuxin Huang', 'DGSRN: Noise-Robust Speech Recognition Method with Dual-Path Gated Spectral Refinement Network', 'wang24da_interspeech', 'addressing noise joint enhancement distortion residue persist issue dense suppression'], ['Gahye Kim, Yunjung Eom, Selina S. Sung, Seunghee Ha, Tae-Jin Yoon, Jungmin So', 'Automatic Children Speech Sound Disorder Detection with Age and Speaker Bias Mitigation', 'kim24q_interspeech', 'ssd debiasing age-dependent impediment dataset group pivotal childhood mitigating multi-head'], ['Jen-Tzung Chien, I-Ping Yeh, Man-Wai Mak', 'Collaborative Contrastive Learning for Hypothesis Domain Adaptation', 'chien24c_interspeech', 'source speaker domain-invariant collaboratively harsh pursue pseudo dual data representation'], ['Xuefei Li, Hao Huang, Ying Hu, Liang He, Jiabao Zhang, Yuyi Wang', 'YOLOPitch: A Time-Frequency Dual-Branch YOLO Model for Pitch Estimation', 'li24ja_interspeech', 'sota determination unvoiced music voiced additional detection f-score proposing accuracy'], ['Qi Wu', 'Mandarin T3 Production by Chinese and Japanese Native Speakers', 'wu24m_interspeech', 'tone adjacent face learner influence challenge underscore study creaky mispronunciation'], ['Saturnino Luz, Sofia De La Fuente Garcia, Fasih Haider, Davida Fromm, Brian MacWhinney, Alyssa Lanzi, Ya-Ning Chang, Chia-Ju Chou, Yi-Chien Liu', 'Connected Speech-Based Cognitive Assessment in Chinese and English', 'luz24_interspeech', 'prediction diagnosis score impairment propensity language-agnostic encompass dataset assess generalise'], ['Jae-Hong Lee, Sang-Eon Lee, Dong-Hyun Kim, DoHee Kim, Joon-Hyuk Chang', 'Online Subloop Search via Uncertainty Quantization for Efficient Test-Time Adaptation', 'lee24j_interspeech', 'iteration leader updated method number quantizes inefficiency test thread sample'], ['Haoyu Wang, Guoqiang Hu, Guodong Lin, Wei-Qiang Zhang, Jian Li', 'Simul-Whisper: Attention-Guided Streaming Whisper with Truncation Detection', 'wang24ea_interspeech', 'chunk decoding chunk-based out-of-distribution hinders cross-attention truncated auto-regressive impressive encoder-decoder'], ['Arnav Goel, Medha Hira, Anubha Gupta', 'Exploring Multilingual Unseen Speaker Emotion Recognition: Leveraging Co-Attention Cues in Multitask Learning', 'goel24_interspeech', 'ser benchmark cross-validation field ravdess crema-d emodb advent -fold wavlm'], ['Ilseok Kim, Ju-Seok Seong, Joon-Hyuk Chang', 'Few-Shot Keyword-Incremental Learning with Total Calibration', 'kim24r_interspeech', 'fill session kw class new prototype keywords loses initial mixup'], ['Amit Meghanani, Thomas Hain', 'LASER: Learning by Aligning Self-supervised Representations of Speech for Improving Content-related Tasks', 'meghanani24_interspeech', 'cost-effective wavlm hubert fine-tuning regularisation ssl-based superb observed continuing asr'], ['Siddique Latif, Raja Jurdak, Björn W. Schuller', 'Evaluating Transformer-Enhanced Deep Reinforcement Learning for Speech Emotion Recognition', 'latif24_interspeech', 'ser rnns transformer-based transformer benchmark speech-emotion prior centred using recent'], ['Eunseop Yoon, Hee Suk Yoon, John Harvill, Mark Hasegawa-Johnson, Chang D. Yoo', 'LI-TTA: Language Informed Test-Time Adaptation for Automatic Speech Recognition', 'yoon24c_interspeech', 'tta self-supervision correction linguistic asr shift exemplification diverges loss wherein'], ['Georgios Paraskevopoulos, Chara Tsoukala, Athanasios Katsamanis, Vassilis Katsouros', 'The Greek podcast corpus: Competitive speech models for low-resourced languages with weakly supervised data', 'paraskevopoulos24_interspeech', 'modern data-intensive silver large-v exacerbated assembling compile podcasts technology correlating'], ['Qiquan Zhang, Hongxu Zhu, Xinyuan Qian, Eliathamby Ambikairajah, Haizhou Li', 'An Exploration of Length Generalization in Transformer-Based Speech Enhancement', 'zhang24n_interspeech', 'transformer position embedding utterance ape explore infeasible unexplored quadratic facilitated'], ['Shareef Babu Kalluri, Prachi Singh, Pratik Roy Chowdhuri, Apoorva Kulkarni, Shikha Baghel, Pradyoth Hegde, Swapnil Sontakke, Deepak K T, S.R. Mahadeva Prasanna, Deepu Vijayasenan, Sriram Ganapathy', 'The Second DISPLACE Challenge: DIarization of SPeaker and LAnguage in Conversational Environments', 'kalluri24_interspeech', 'dataset track hour leader recording asr board far-field highlighted baseline'], ['Marc Frei
11146xes, Marc Arnela, Joan Claudi Socoró, Luis Joglar-Ongay, Oriol Guasch, Francesc Alías-Pujol', 'Glottal inverse filtering and vocal tract tuning for the numerical simulation of vowel /a/ with different levels of vocal effort', 'freixes24_interspeech', 'source low methodology predominates inverse-filtered liljencrants-fant adjusts model validates reproducing'], ['Shuochen Gao, Shun Lei, Fan Zhuo, Hangyu Liu, Feng Liu, Boshi Tang, Qiaochu Huang, Shiyin Kang, Zhiyong Wu', 'An End-to-End Approach for Chord-Conditioned Song Generation', 'gao24e_interspeech', 'chord music accompaniment flaw vocal harmony inaccuracy control cross-attention lyric'], ['Shuai Wang, Ke Zhang, Shaoxiong Lin, Junjie Li, Xuefei Wang, Meng Ge, Jianwei Yu, Yanmin Qian, Haizhou Li', 'WeSep: A Scalable and Flexible Toolkit Towards Generalizable Target Speaker Extraction', 'wang24fa_interspeech', 'tse avaliable subsequential isolating featured toolkits off-the-shelf cocktail multi-talker on-the-fly'], ['Minmin Yang, Rachid Ridouane', 'Intrusive schwa within French stop-liquid clusters: An acoustic analysis', 'yang24n_interspeech', 'word-final liquid consonant stop occurrence place articulation position prevalence factor'], ['Wenhao Guan, Kaidi Wang, Wangjin Zhou, Yang Wang, Feng Deng, Hui Wang, Lin Li, Qingyang Hong, Yong Qin', 'LAFMA: A Latent Flow Matching Model for Text-to-Audio Generation', 'guan24b_interspeech', 'diffusion audio step regressing tta sample generated space sacrificing facilitated'], ['Xiaopeng Wang, Ruibo Fu, Zhengqi Wen, Zhiyong Wang, Yuankun Xie, Yukun Liu, Jianhua Tao, Xuefei Liu, Yongwei Li, Xin Qi, Yi Lu, Shuchen Shi', 'Genuine-Focused Learning using Mask AutoEncoder for Generalized Fake Audio Detection', 'wang24ga_interspeech', 'fad genuine mae spoofed reconstruction feature gfl content-related focus supplement'], ['Xueyuan Chen, Dongchao Yang, Dingdong Wang, Xixin Wu, Zhiyong Wu, Helen Meng', 'CoLM-DSR: Leveraging Neural Codec Language Modeling for Multi-Modal Dysarthric Speech Reconstruction', 'chen24t_interspeech', 'prosody dsr naturalness codecs similarity speaker embeddings encoder normal extract'], ['Peikun Chen, Sining Sun, Changhao Shan, Qing Yang, Lei Xie', 'Streaming Decoder-Only Automatic Speech Recognition with Discrete Speech Units: A Pilot Study', 'chen24u_interspeech', 'non-streaming unified token model attention introduce speech-text designed asr speech-related'], ['Valentin Pelloin, Léna Dodson, Émile Chapuis, Nicolas Hervé, David Doukhan', 'Automatic Classification of News Subjects in Broadcast News: Application to a Gender Bias Representation Analysis', 'pelloin24_interspeech', 'topic woman llm dataset politics french channel broadcasted delineate computational'], ['Xuanru Zhou, Anshul Kashyap, Steve Li, Ayati Sharma, Brittany Morin, David Baquirin, Jet Vonk, Zoe Ezzes, Zachary Miller, Maria Tempini, Jiachen Lian, Gopala Anumanchipalli', 'YOLO-Stutter: End-to-end Region-Wise Speech Dysfluency Detection', 'zhou24e_interspeech', 'dysfluencies dysfluent aggregator aphasia open-sourced speech-text prolongation governed imperfect state-of-the-art'], ['Ziqian Ning, Shuai Wang, Pengcheng Zhu, Zhichao Wang, Jixun Yao, Lei Xie, Mengxiao Bi', 'DualVC 3: Leveraging Language Model Generated Pseudo Context for End-to-end Low Latency Streaming Voice Conversion', 'ning24_interspeech', 'encoder optional multi-level popularity k-means cascade adopts chunk clustered ssl'], ['Chao-Wei Huang, Hui Lu, Hongyu Gong, Hirofumi Inaguma, Ilia Kulikov, Ruslan Mavlyutov, Sravya Popuri', 'Investigating Decoder-only Large Language Models for Speech-to-text Translation', 'huang24h_interspeech', 'llm covost fleurs consume parameter-efficient exceptional proprietary avenue task speech-related'], ['Dake Guo, Xinfa Zhu, Liumeng Xue, Yongmao Zhang, Wenjie Tian, Lei Xie', 'Text-aware and Context-aware Expressive Audiobook Speech Synthesis', 'guo24d_interspeech', 'style diverse tt expressiveness encoder vits-based capture narrator space audiobooks'], ['Hang Zhao, Yifei Xin, Zhesong Yu, Bilei Zhu, Lu Lu, Zejun Ma', 'MINT: Boosting Audio-Language Model via Multi-Target Pre-Training and Instruction Tuning', 'zhao24h_interspeech', 'audio-text frozen task diverse integration alignment empowers generation cross-modality understanding'], ['Yudong Yang, Rongfeng Su, Rukiye Ruzi, Manwa Ng, Shaofeng Zhao, Nan Yan, Lan Wang', 'Optical Flow Guided Tongue Trajectory Generation for Diffusion-based Acoustic to Articulatory Inversion', 'yang24o_interspeech', 'uti aai diffusion reference generated data omitting constraint contour additional'], ['Xiang-Li Lu, Yi-
11146Fen Liu', 'Deep Prosodic Features in Tandem with Perceptual Judgments of Word Reduction for Tone Recognition in Conversed Speech', 'lu24d_interspeech', 'classification transformer-based rhythmic tackle leveraging encoding classify predicting jointly encoder'], ['Han Kunmei', 'Modelling Lexical Characteristics of the Healthy Aging Population: A Corpus-Based Study', 'kunmei24_interspeech', 'nlp language old age concreteness large-language prodromal tool help normative'], ['Shaojun Li, Daimeng Wei, Hengchao Shang, Jiaxin Guo, ZongYao Li, Zhanglin Wu, Zhiqiang Rao, Yuanchang Luo, Xianghui He, Hao Yang', 'Speaker-Smoothed kNN Speaker Adaptation for End-to-End ASR', 'li24ka_interspeech', 'pre-built setting k-nearest comparably data neighbor x-vector sparsity cer adjust'], ['Jeong-Hwan Choi, Ye-Rin Jeoung, Ilseok Kim, Joon-Hyuk Chang', 'Efficient Speaker Embedding Extraction Using a Twofold Sliding Window Algorithm for Speaker Diarization', 'choi24d_interspeech', 'frame-level representation concatenated pre-trained employ extract floating-point adapter non-overlapping distillation'], ['Mun-Hak Lee, Jae-Hong Lee, DoHee Kim, Ye-Eun Ko, Joon-Hyuk Chang', 'Balanced-Wav2Vec: Enhancing Stability and Robustness of Representation Learning Through Sample Reweighting Techniques', 'lee24k_interspeech', 'collapse mode wav codebook exacerbates stably loss skewed suppresses converges'], ['Xuankai Chang, Jiatong Shi, Jinchuan Tian, Yuning Wu, Yuxun Tang, Yihan Wu, Shinji Watanabe, Yossi Adi, Xie Chen, Qin Jin', 'The Interspeech 2024 Challenge on Speech Processing Using Discrete Units', 'chang24b_interspeech', 'compelling assess pivotal encompasses foster evolving restoration singing highlighted baseline'], ['Hang Su, Yuxiang Kong, Lichun Fan, Peng Gao, Yujun Wang, Zhiyong Wu', 'Speaker Change Detection with Weighted-sum Knowledge Distillation based on Self-supervised Pre-trained Models', 'su24_interspeech', 'scd method fine-tuning model basic consumes many industrial selectively industry'], ['Yeh-Sheng Lin, Shu-Chuan Tseng, Jyh-Shing Roger Jang', 'Leveraging Phonemic Transcription and Whisper toward Clinically Significant Indices for Automatic Child Speech Assessment', 'lin24k_interspeech', 'normative diagnosing early speech-language ssd unsuitable workflow phoneme validates pathologist'], ['Rui Wang, Liping Chen, Kong Aik Lee, Zhen-Hua Ling', 'Asynchronous Voice Anonymization Using Adversarial Perturbation On Speaker Embedding', 'wang24ha_interspeech', 'attribute perception human preserved pseudo-speaker obscured machine anonymized disentanglement altering'], ['Yishuang Li, Wenhao Guan, Hukai Huang, Shiyu Miao, Qi Su, Lin Li, Qingyang Hong', 'Efficient Integrated Features Based on Pre-trained Models for Speaker Verification', 'li24la_interspeech', 'ptms handcrafted representation undoubtedly decent discarded fine-tune adaptively fuse multi-layer'], ['Jiu Feng, Mehmet Hamza Erol, Joon Son Chung, Arda Senocak', 'ElasticAST: An Audio Spectrogram Transformer for All Length and Resolutions', 'feng24c_interspeech', 'asts ast inference flexibility input packing inherit accommodates fixed-size training'], ['Kun Zou, Fengyun Tan, Ziyang Zhuang, Chenfeng Miao, Tao Wei, Shaodan Zhai, Zijian Li, Wei Hu, Shaojun Wang, Jing Xiao', 'E-Paraformer: A Faster and Better  Parallel Transformer for Non-autoregressive End-to-End Mandarin Speech Recognition', 'zou24_interspeech', 'paraformer cif speedup aishell- embeddings integrate-and-fire inefficiency nar mechanism token-level'], ['Yu Watanabe, Koichiro Ito, Shigeki Matsubara', 'Utilization of Text Data for Response Timing Detection in Attentive Listening', 'watanabe24_interspeech', 'narrative punctuation agent utilize accumulated trained model insertion mark inspired'], ['Ziyun Cui, Chang Lei, Wen Wu, Yinan Duan, Diyang Qu, Ji Wu, Runsen Chen, Chao Zhang', 'Spontaneous Speech-Based Suicide Risk Detection Using Whisper and Large Language Models', 'cui24_interspeech', 'adolescent finetuning llm potential parameter-efficient eighteen audio-text speech -score aged'], ['Chia-Kai Yeh, Chih-Chun Chen, Ching-Hsien Hsu, Jen-Tzung Chien', 'Cross-Modality Diffusion Modeling and Sampling for Speech Recognition', 'yeh24_interspeech', 'discrete transformer excels decorrelation continuous serving non-autoregressive merit identifies redundancy'], ['Ainikaerjiang Aimaiti, Di Wu, Liting Jiang, Gulinigeer Abudouwaili, Hao Huang, Wushour Silamu', 'An Uyghur Extension to the MASSIVE Multi-lingual Spoken Language Understanding Corpus with Comprehensive Evaluations', 'aimaiti24_interspeech', 'slu dataset embedding affix agglutinative available stem task-oriented com github'], ['Murali Karthick Baskar, Andrew Rosenberg, Bhuvana Ramabhadran, Neeraj Gaur, Zhong Meng', 'Speech Prefix-Tuning with RNNT Loss for Improving LLM Predictions', 'baskar24_interspeech', 'frozen indic prefix fine-tuned asr applying approches testset language-based lead'], ['Mojtaba Kadkhodaie Elyaderani, John Glover, Thomas Schaaf', 'Reference-Free Estimation of the Quality of Clinical Notes Generated from Doctor-Patient Conversations', 'kadkhodaieelyaderani24_interspeech', 'n
11146ote quot collection automatically-generated reference-based generation robust section counterpart absence'], ['Tianqi Geng, Hui Feng', "Form and Function in Prosodic Representation:  In the Case of 'ma' in Tianjin Mandarin", 'geng24_interspeech', 'deixis pronoun particle modal marker pitch discourse distinct narrowest widest'], ['Shijie Lai, Minglu He, Zijing Zhao, Kai Wang, Hao Huang, Jichen Yang', 'Synthesizing Long-Form Speech merely from Sentence-Level Corpus with Content Extrapolation and LLM Contextual Enrichment', 'lai24b_interspeech', 'generalization pause llm-based generate model natural equipped gated failure fail'], ['Bingliang Zhao, Jiangping Kong, Xiyu Wu', 'Age-related Differences in Acoustic Cues for the Perception of Checked Syllables in Shengzhou Wu', 'zhao24i_interspeech', 'coda stop glottal evolution perceptual vowel duration chinese unchecked weaken'], ['Sreyan Ghosh, Sonal Kumar, Ashish Seth, Purva Chiniya, Utkarsh Tyagi, Ramani Duraiswami, Dinesh Manocha', 'LipGER: Visually-Conditioned Generative Error Correction for Robust Automatic Speech Recognition', 'ghosh24b_interspeech', 'motion lip asr llm visual cue instruct datasets beam-search avsr'], ['David Doukhan, Lena Dodson, Manon Conan, Valentin Pelloin, Aurélien Clamouse, Mélina Lepape, Géraldine Van Hille, Cécile Méadel, Marlène Coulomb-Gully', 'Gender Representation in TV and Radio: Automatic Information Extraction methods versus Manual Analyses', 'doukhan24_interspeech', 'woman descriptor channel protagonist french war reference systemic depicting journalist'], ['Neeraj Gaur, Rohan Agrawal, Gary Wang, Parisa Haghani, Andrew Rosenberg, Bhuvana Ramabhadran', 'ASTRA: Aligning Speech and Text Representations for Asr without Sampling', 'gaur24_interspeech', 'rnnt match modality length fleurs duration-based injection prevailing upsampling misalignment'], ['Anaïs Rameau, Satrajit Ghosh, Alexandros Sigaras, Olivier Elemento, Jean-Christophe Belisle-Pipon, Vardit Ravitsky, Maria Powell, Alistair Johnson, David Dorr, Philip Payne, Micah Boyer, Stephanie Watts, Ruth Bahr, Frank Rudzicz, Jordan Lerner-Ellis, Shaheen Awan, Don Bolser, Yael Bensoussan', 'Developing Multi-Disorder Voice Protocols: A team science approach involving clinical expertise, bioethics, standards, and DEI.', 'rameau24_interspeech', 'respiratory cohort disease large-scale ethically fuel biomarkers sourced pro anxiety'], ['Yuewei Zhang, Huanbin Zou, Jie Zhu', 'Sub-PNWR: Speech Enhancement Based on Signal Sub-Band Splitting and Pseudo Noisy Waveform Reconstruction Loss', 'zhang24o_interspeech', 'full-band denoised bank filter module spectrum computational spaced devise entail'], ['Kai-Wei Chang, Ming-Hao Hsu, Shan-Wen Li, Hung-yi Lee', 'Exploring In-Context Learning of Textless Speech Language Model for Speech Classification Tasks', 'chang24c_interspeech', 'icl demonstration nlp manner equipping capability gpt- first perform few-shot'], ['Akihiro Kato, Hiroyuki Nagano, Kohei Chike, Masaki Nose', 'Self-Supervised Learning for ASR Pre-Training with Uniquely Determined Target Labels and Controlling Cepstrum Truncation for Speech Augmentation', 'kato24_interspeech', 'pre-trained libri-light limited hubert data conformer method condition cer non'], ['Fangjing Niu, Xiaozhe Qi, Xinya Chen, Liang He', 'Speech Topic Classification Based on Multi-Scale and Graph Attention Networks', 'niu24b_interspeech', 'node transcribed convolutional global capture relationship stc surpassing subjected text'], ['Jinchuan Tian, Yifan Peng, William Chen, Kwanghee Choi, Karen Livescu, Shinji Watanabe', 'On the Effects of Heterogeneous Data Sources on Speech-to-Text Foundation Models', 'tian24_interspeech', 'owsm open series staying heterogeneity transparency proxy less punctuation llm'], ['Fathima Zaheera, Supritha Shetty, Gayadhar Pradhan, Deepak K T', 'Automatic Assessment of Dysarthria using Speech and synthetically generated Electroglottograph signal', 'zaheera24_interspeech', 'lmfe dysarthric stm averaged mfccs mel complementary automated computed ua-speech'], ['Yixiang Niu, Ning Chen, Hongqing Zhu, Zhiying Zhu, Guangqiang Li, Yibo Chen', 'Auditory Spatial Attention Detection Based on Feature Disentanglement and Brain Connectivity-Informed Graph Neural Networks', 'niu24c_interspeech', 'eeg asad connectivity cross-subject single-trial surround inter-subject electroencephalogram graph-based model'], ['Tianhua Qi, Shiyan Wang, Cheng Lu, Yan Zhao, Yuan Zong, Wenming Zheng', 'Towards Realistic Emotional Voice Conversion using Controllable Emotional Intensity', 'qi24_interspeech', 'evc diversity rhythm emotion renderer pseudo-labels mere smoothly controllability feature'], ['Jincen Wang, Yan Zhao, Cheng Lu, Chuangao Tang, Sunan Li, Yuan Zong, Wenming Zheng', 'Boosting Cross-Corpus Speech Emotion Recognition using CycleGAN with Contrastive Learning', 'wang24ia_interspeech', 'premise gan ser data casia testing enterface synthetic originates identically'], ['Cheng Lu, Yuan Zong, Yan Zhao, Hailun Lian, Tianhua Qi, Björn Schuller, Wenming Zheng', 'Hierarchical Distribution Adaptation for Unsupervised Cross-corpus Speech Emotion Recognition', 'lu24e_interspeech', 'hda testing utterance-level ser domain undermines sda uda module fda'], ['Jacob Bitterman, Daniel Levi, Hilel Hagai Diamandi, Sharon Gannot, Tal Rosenwein', 'RevRIR: Joint Reverberant Speech and Room Impulse Response Embedding using Contrastive Learning with Application to Room Shape Classification', 'bitterman24_interspeech', 'rir task fingerprinting determine cumbersome specific embed receives volume utterance'], ['Tianyi Xu, Kaixun Huang, Pengcheng Guo, Yu Zhou, Longtao Huang, Hui Xue, Lei Xie', 'Towards Rehearsal-Free Multilingual ASR: A LoRA-based Case Study on Whisper ', 'xu24h_interspeech', 'forgetting original language model parameter uyghur tibetan allocate new lora'], ['Jincen Wang, Yan Zhao, Cheng Lu, Hailun Lian, Hongli Chang, Yuan Zong, Wenming Zheng', 'Confidence-aware Hypothesis Transfer Networks for Source-Free Cross-Corpus Speech Emotion Recognition', 'wang24ja_interspeech', 'ser source module target emovo casia enterface emodb self-training confident'], ['Ofer Schwartz, Sharon Gannot', 'Efficient Joint Bemforming and Acoustic Echo Cancellation Structure for Conference Call Scenarios', 'schwartz24_interspeech', 'aec microphone scheme pre-filter indifferent weight apply far-end cir
11146cumvent change'], ['Zhu Li, Xiyuan Gao, Yuqing Zhang, Shekhar Nayak, Matt Coler', 'A Functional Trade-off between Prosodic and Semantic Cues in Conveying Sarcasm', 'li24ma_interspeech', 'sarcastic phrase semantics meaning expression utterance propensity illocutionary disentangles propositional'], ['Katerina Papadimitriou, Gerasimos Potamianos', 'Multimodal Continuous Fingerspelling Recognition via Visual Alignment Learning', 'papadimitriou24_interspeech', 'skeletal signing st-gcn bigru d-cnn paramount spatio-temporal persist accessibility parameterization'], ['Sathvik Udupa, Soumi Maiti, Prasanta Kumar Ghosh', 'IndicMOS: Multilingual MOS Prediction for 7 Indian languages', 'udupa24b_interspeech', 'tt predictor level open-source evaluation indic train assess challenge gold'], ['Xiuwen Zheng, Bornali Phukon, Mark Hasegawa-Johnson', "Fine-Tuning Automatic Speech Recognition for People with Parkinson's: An Effective Strategy for Enhancing Speech Technology Accessibility", 'zheng24c_interspeech', 'multi-task librispeech sap palsy model dysphonic cerebral asr dysarthric data'], ['Peter Birkholz, Patrick Häsner', 'Measurement and simulation of pressure losses due to airflow in vocal tract models', 'birkholz24_interspeech', 'tube viscous section proportional glottis kinetic power area diameter bernoulli'], ['Khanh Le, Duc Chau', 'Improving Streaming Speech Recognition With Time-Shifted Contextual Attention And Dynamic Right Context Masking', 'le24_interspeech', 'tsca user-perceived drc future in-context chunk-based valued restricts featuring stand'], ['Vu Hoang, Viet Thanh Pham, Hoa Nguyen Xuan, Pham Nhi, Phuong Dat, Thi Thu Trang Nguyen', 'VSASV: a Vietnamese Dataset for Spoofing-Aware Speaker Verification', 'hoang24b_interspeech', 'sasv spoofed language system authentic unexplored anti-spoofing concentrated encourage latest'], ['Tuyen Tran, Khanh Le, Ngoc Dang Nguyen, Minh Vu, Huyen Ngo, Woomyoung Park, Thi Thu Trang Nguyen', 'VN-SLU: A Vietnamese Spoken Language Understanding Dataset', 'tran24b_interspeech', 'slu ensuring intent tool crowd-sourcing smart slot scarcity home utterance'], ['Hemant Yadav, Sunayana Sitaram, Rajiv Ratn Shah', 'MS-HuBERT: Mitigating Pre-training and Inference Mismatch in Masked Language Modelling methods for learning Speech Representations', 'yadav24b_interspeech', 'hubert self-supervised asr swap traction disparity finetuning beat vanilla lag'], ['Mattias Nilsson, Riccardo Miccini, Clement Laroche, Tobias Piechowiak, Friedemann Zenke', 'Resource-Efficient Speech Quality Prediction through Quantization Aware Training and Binary Activation Maps', 'nilsson24_interspeech', 'mixed-precision bam device dnsmos commonplace dot prohibitive resource-constrained multiplication -bit'], ['Atli Sigurgeirsson, Eddie L. Ungless', "Just Because We Camp, Doesn't Mean We Should: The Ethics of Modelling Queer Voices.", 'sigurgeirsson24_interspeech', 'gay voice pipeline rating lgbtq ramification death capture fairness loss'], ['Adham Ibrahim, Shady Shehata, Ajinkya Kulkarni, Mukhtar Mohamed, Muhammad Abdul-Mageed', 'What Does it Take to Generalize SER Model Across Datasets? A Comprehensive Benchmark', 'ibrahim24_interspeech', 'emotional emotion oversampling explore imbalanced emphasizing thorough whisper speech-based evaluation'], ['Alexander Blatt, Aravind Krishnan, Dietrich Klakow', 'Joint vs Sequential Speaker-Role Detection and Automatic Speech Recognition for Air-traffic Control', 'blatt24_interspeech', 'srd atc architecture asr transcript atco natural-language traditional step preferable'], ['Weiqin Li, Peiji Yang, Yicheng Zhong, Yixuan Zhou, Zhisheng Wang, Zhiyong Wu, Xixin Wu, Helen Meng', 'Spontaneous Style Text-to-Speech Synthesis with Controllable Spontaneous Behaviors Based on Language Models', 'li24na_interspeech', 'prosody speech diverse naturalness low-quality variation categorize uniformly human-like encounter'], ['Sneha Ray Barman, Shakuntala Mahanta, Neeraj Kumar Sharma', 'Deciphering Assamese Vowel Harmony with Featural InfoWaveGAN', 'raybarman24_interspeech', 'regressive long-distance iterative learning illicit grasping non-iterative insightful curated universality'], ['Orchid Chetia Phukan, Priyabrata Mallick, Swarup Ranjan Behera, Aalekhya Satya Narayani, Arun Balaji Buduru, Rajesh Sharma', 'Towards Multilingual Audio-Visual Question Answering', 'phukan24_interspeech', 'avqa benchmark datasets work existing replicating language suite prevents allocation'], ['Hongchen Wu, Jiwon Yun', 'Influences of Morphosyntax and Semantics on the Intonation of Mandarin Chinese Wh-indeterminates', 'wu24n_interspeech', 'morphosyntactic production faculty prosodic shaping interrogative clause convey speech interact'], ['Jinhyeok Yang, Junhyeok Lee, Hyeong-Seok Choi, Seunghoon Ji, Hyeongju Kim, Juheon Lee', 'DualSpeech: Enhancing Speaker-Fidelity and Text-Intelligibility Through Dual Classifier-Free Guidance', 'yang24q_interspeech', 'tt control exceptional nuance diffusion phoneme-level replicate surpasses demo advancement'], ['Tianchi Liu, Lin Zhang, Rohan Kumar Das, Yi Ma, Ruijie Tao, Haizhou Li', 'How Do Neural Spoofing Countermeasures Detect Partially Spoofed Audio?', 'liu24m_interspeech', 'cm bona fide prioritize decision-making lay interpretability focus explains manipulating'], ['Tillmann Pistor, Adrian Leemann', 'Echoes of Implicit Bias Exploring Aesthetics and Social Meanings of Swiss German Dialect Features', 'pistor24_interspeech', 'bern zurich stereotypical aesthetic importance category stereotype single-word rater perception'], ['Aryan Chaudhary, Vinayak Abrol', 'QGAN: Low Footprint Quaternion Neural Vocoder for Speech Synthesis', 'chaudhary24b_interspeech', 'low-footprint vastly landscape model grown gans evolved compromising diffusion high-fidelity'], ['Edresson Casanova, Kelly Davis, Eren Gölge, Görkem Göknar, Iulian Gulea, Logan Hart, Aya Aljafari, Joshua Meyer, Reuben Morais, Samuel Olayemi, Julian Weber', 'XTTS: a Massively Multilingual Zero-Shot Text-to-Speech Model', 'casanova24_interspeech', 'zs-tts medium language voicebox vall-e resource cloning sota limiting proposing'], ['Nan Chen, Yonghe Wang, Feilong Bao', 'Sign Value Constraint Decomposition for Efficient 1-Bit Quantization of Speech Translation Tasks', 'chen24v_interspeech', 'quantized model matrix layer cost linear approximates distillation speech-to-text trainable'], ['Devang Kulshreshtha, Nikolaos Pappas, Brady Houston, Saket Dingliwal, Srikanth Ronanki', 'Sequential Editing for Lifelong Training of Speech Recognition Models', 'kulshreshtha24_interspeech', 'domain lll necessitate prior new asr fine-tuning access multi-accent inefficiency'], ['Vincenzo Norman Vitale, Loredana Schettino, Francesco Cutugno', 'Rich speech signal: exploring and exploiting  end-to-e
11146nd automatic speech recognizers’ ability to model hesitation phenomena', 'vitale24_interspeech', 'decoder disrupting deepen disregarded procedural system prolongation conformer-based neglect acknowledged'], ['Anna Favaro, Tianyu Cao, Najim Dehak, Laureano Moro-Velazquez', 'Leveraging Universal Speech Representations for Detecting and Assessing the Severity of Mild Cognitive Impairment Across Languages', 'favaro24_interspeech', 'mci feature mmse taukadial cross-lingually language-agnostic mini-mental interpretable suitability cohort'], ['Zitha Sasindran, Harsha Yelchuri, T. V. Prabhakar', 'SeMaScore: A new evaluation metric for automatic speech recognition tasks', 'sasindran24_interspeech', 'bertscore serf segment-wise score atypical corresponds algorithm signal-to-noise scoring involving'], ['Bhasi K. C., Rajeev Rajan, Noumida A', 'Attention-augmented X-vectors for the Evaluation of Mimicked Speech Using Sparse Autoencoder-LSTM framework', 'kc24_interspeech', 'artist mimicking fed encoded later embeddings rank- mimicry competency lstm-based'], ['Vidar Freyr Gudmundsson, Keve Márton Gönczi, Malin Svensson Lundmark, Donna Erickson, Oliver Niebuhr', 'The MARRYS helmet: A new device for researching and training “jaw dancing”', 'gudmundsson24_interspeech', 'portrait articulograph ema electromagnetic teaching motivation outline science illustrate analyzing'], ['Nikhil Jakhar, Sudhanshu Srivastava, Arun Baby', 'A Unified Approach to Multilingual Automatic Speech Recognition with Improved Language Identification for Indic Languages', 'jakhar24_interspeech', 'lid asr scalability der whisper diarization model language-specific low-resource enhancing'], ['Patrick Cormac English, John D. Kelleher, Julie Carson-Berndsen', 'Searching for Structure: Appraising the Organisation of Speech Features in wav2vec 2.0 Embeddings', 'english24_interspeech', 'within transformer organisational relationship encapsulate uncovering explainability model probing mining'], ['Danilo de Oliveira, Simon Welker, Julius Richter, Timo Gerkmann', "The PESQetarian: On the Relevance of Goodhart's Law for Speech Enhancement", 'deoliveira24_interspeech', 'metric pesq model evaluation listening misleading detrimental imply instrumental used'], ['Liam Kelley, Diego Di Carlo, Aditya Arie Nugraha, Mathieu Fontaine, Yoshiaki Bando, Kazuyoshi Yoshii', 'RIR-in-a-Box: Estimating Room Acoustics from 3D Mesh Data through Shoebox Approximation', 'kelley24_interspeech', 'rirs rir latent differentiable code scene real estimator consistency physical'], ['Himanshu Maurya, Atli Sigurgeirsson', 'A Human-in-the-Loop Approach to Improving Cross-Text Prosody Transfer', 'maurya24_interspeech', 'reference hitl text rendition target prosodic utterance appropriate suffices closeness'], ['Themos Stafylakis, Anna Silnova, Johan Rohdin, Oldřich Plchot, Lukáš Burget', 'Challenging margin-based speaker embedding extractors by using the variational information bottleneck', 'stafylakis24_interspeech', 'loss margin logit principled angular merely softmax target deterministic cross-entropy'], ['Federico Lo Iacono, Valentina Colonna, Antonio Romano', 'Preservation, conservation and phonetic study of the voices of Italian poets: A study on the seven years of the VIP archive', 'loiacono24_interspeech', 'poetic cultural preserving oral crucial conserving inheritance fragile project culturally'], ['Si-Ioi Ng, Lingfeng Xu, Kimberly D. Mueller, Julie Liss, Visar Berisha', 'Segmental and Suprasegmental Speech Foundation Models for Classifying Cognitive Risk Factors: Evaluating Out-of-the-Box Performance', 'ng24_interspeech', 'trillsson clinical mci dementia precludes macro-f intraclass use-cases analysis challenged'], ['Miku Nishihara, Dan Wells, Korin Richmond, Aidan Pine', 'Low-dimensional Style Token Control for Hyperarticulated Speech Synthesis', 'nishihara24_interspeech', 'clear pca variation gsts gst dimension tt example space text-tospeech'], ['Janek Ebbers, François G. Germain, Gordon Wichern, Jonathan Le Roux', 'Sound Event Bounding Boxes', 'ebbers24_interspeech', 'extent presence frame-level confidence thresholding prediction posterior psds frame decouple'], ['Youssef Nafea, Shady Shehata, Zeerak Talat, Ahmed Aboeitta, Ahmed Sharshar, Preslav Nakov', 'AraOffence: Detecting Offensive Speech Across Dialects in Arabic Media', 'nafea24_interspeech', 'nlp labelled text image trail dialectical abuse under-represented content swahili'], ['Yuri Khokhlov, Tatiana Prisyach, Anton Mitrofanov, Dmitry Dutov, Igor Agafonov, Tatiana Timofeeva, Aleksei Romanenko, Maxim Korenevsky', 'Classification of Room Impulse Responses and its application for channel verification and diarization', 'khokhlov24_interspeech', 'rir embeddings rirs libricss acoustic reverb bearing experiment voxceleb reasonable'], ['Alessandro De Luca, Andrew Clark, Volker Dellwo', 'NumberLie: a game-based experiment to understand the acoustics of deception and truthfulness', 'deluca24_interspeech', 'player immediate truth game precise consequence backed trustworthiness lying tailor'], ['Rongshuai Wu, Debasish Ray Mohapatra, Sidney Fels', 'Modeling Vocal Tract Like Acoustic Tubes Using the Immersed Boundary Method', 'wu24o_interspeech', 'solver wave fdtd tube less grid geometry regular lagrangian finite-difference'], ['Johanna Cronenberg, Ioana Chitoran, Lori Lamel, Ioana Vasilescu', 'Crosslinguistic Comparison of Acoustic Variation in the Vowel Sequences /ia/ and /io/ in Four Romance Languages', 'cronenberg24_interspeech', 'hiatus di
11146phthong differ formant prefers respect account romanian mix inclusion'], ['Aleksei Gusev, Anastasia Avdeeva', 'Improvement Speaker Similarity for Zero-Shot Any-to-Any Voice Conversion of Whispered and Regular Speech', 'gusev24_interspeech', 'transfer generated unexplored domain lightweight truth streaming quality ground reconstruct'], ['Haibin Wu, Yuan Tseng, Hung-yi Lee', 'CodecFake: Enhancing Anti-Spoofing Models Against Deepfake Audios from Codec-Based Speech Synthesis Systems', 'wu24p_interspeech', 'sota counter current detect synthesized unanswered empowers effectively misuse curate'], ['Joseph Coffey, Okko Räsänen, Camila Scaff, Alejandrina Cristia', 'The Difficulty and Importance of Estimating the Lower and Upper Bounds of Infant Speech Exposure', 'coffey24_interspeech', 'estimate unanimous year computational range plausibility language benchmarking discussing well-established'], ['John Murzaku, Adil Soubki, Owen Rambow', 'Multimodal Belief Prediction', 'murzaku24_interspeech', 'bert whisper audio corpus fine-tunes commitment text-only approached task text'], ['Dominik Wagner, Ilja Baumann, Korbinian Riedhammer, Tobias Bocklet', 'Outlier Reduction with Gated Attention for Improved Post-training Quantization in Large Sequence-to-sequence Speech Foundation Models', 'wagner24_interspeech', 'gating whisper transformer-based student impede necessitating mechanism tensor mitigation -bit'], ["Allahsera Tapo, Éric Le Ferrand, Zoey Liu, Christopher Homan, Emily Prud'hommeaux", 'Leveraging Speech Data Diversity to Document Indigenous Heritage and Culture', 'tapo24_interspeech', 'asr history bambara mali mande corpus fieldwork culturally commonality archival'], ['Badr M. Abdullah, Mohammed Maqsood Shaik, Dietrich Klakow', 'Wave to Interlingua: Analyzing Representations of Multilingual Speech Transformers for Spoken Language Translation', 'abdullah24_interspeech', 'encoder shared probing across finding interpretability untranscribed speech-to-text trained transformer-based'], ['Mingshuai Liu, Zhuangqi Chen, Xiaopeng Yan, Yuanjun Lv, Xianjun Xia, Chuanzeng Huang, Yijian Xiao, Lei Xie', 'RaD-Net 2: A causal two-stage repairing and denoising speech enhancement network with knowledge distillation and complex axial self-attention', 'liu24n_interspeech', 'ssi icassp future upgraded dnsmos stage non-causal use challenge receptive'], ['Paula Andrea Pérez-Toro, Tomas Arias-Vergara, Philipp Klumpp, Tobias Weise, Maria Schuster, Elmar Noeth, Juan Rafael Orozco-Arroyave, Andreas Maier', 'Multilingual Speech and Language Analysis for the Assessment of Mild Cognitive Impairment: Outcomes from the Taukadial Challenge', 'pereztoro24_interspeech', 'language-dependent timing hallmark alzheimer dementia neurological uar english manifest decline'], ['Ivan Yakovlev, Rostislav Makarov, Andrei Balykin, Pavel Malov, Anton Okhotnikov, Nikita Torgashov', 'Reshape Dimensions Network for Speaker Recognition', 'yakovlev24_interspeech', 'redimnet block map reshaping aggregation versa vice representation facilitating scalable'], ['Matthew Maciejewski, Dominik Klement, Ruizhe Huang, Matthew Wiesner, Sanjeev Khudanpur', 'Evaluating the Santa Barbara Corpus: Challenges of the Breadth of Conversational Spoken Language', 'maciejewski24_interspeech', 'party speech problem matured overlooking dinner necessitates heterogeneity technology push'], ['Dominik Wagner, Sebastian P. Bayerl, Ilja Baumann, Elmar Noeth, Korbinian Riedhammer, Tobias Bocklet', 'Large Language Models for Dysfluency Detection in Stuttered Speech', 'wagner24b_interspeech', 'multi-label llm dysfluencies finetune stuttering inclusive non-lexical audio processor deployment'], ["Johannah O'Mahony, Catherine Lai, Éva Székely", 'Well, what can you do with messy data? Exploring the prosody and pragmatic function of the discourse marker &quot;well&quot; with found data and speech synthesis', 'omahony24_interspeech', 'realisation prosodic conversational explore controllable synthesise unlabelled centroid untranscribed subtle'], ['Zhongweiyang Xu, Ali Aroudi, Ke Tan, Ashutosh Pandey, Jung-Suk Lee, Buye Xu, Francesco Nesta', 'FoVNet: Configurable Field-of-View Speech Enhancement with Low Computation and Distortion for Smart Glasses', 'xu24i_interspeech', 'multi-channel efficiency ultra-low 
11146solution excels efficient computational designed needing within'], ['Ilja Baumann, Nicole Unger, Dominik Wagner, Korbinian Riedhammer, Tobias Bocklet', 'Automatic Evaluation of a Sentence Memory Test for Preschool Children', 'baumann24_interspeech', 'correctness recited early assessment syntactic capability semantic potential timely childhood'], ['Jinzuomu Zhong, Yang Li, Hui Huang, Korin Richmond, Jie Liu, Zhiba Su, Jing Guo, Benlai Tang, Fengjie Zhu', 'Multi-Modal Automatic Prosody Annotation with Contrastive Pretraining of Speech-Silence and Word-Punctuation', 'zhong24c_interspeech', 'prosodic text-speech stage labor-intensive boundary controllable controllability bearing inconsistent sota'], ['Ilja Baumann, Dominik Wagner, Maria Schuster, Korbinian Riedhammer, Elmar Noeth, Tobias Bocklet', 'Towards Self-Attention Understanding for Automatic Articulatory Processes Analysis in Cleft Lip and Palate Speech', 'baumann24b_interspeech', 'clp phoneme interpretability paving diagnostics anomaly multi-head classification long-range characteristic'], ['Pooneh Mousavi, Jarod Duret, Salah Zaiem, Luca Della Libera, Artem Ploujnikov, Cem Subakan, Mirco Ravanelli', 'How Should We Extract Discrete Audio Tokens from Self-Supervised Models?', 'mousavi24_interspeech', 'ssl semantic ideal configuration optimal layer identify attention benchmarked tokenization'], ['Dominika Woszczyk, Ranya Aloufi, Soteris Demetriou', 'Prosody-Driven Privacy-Preserving Dementia Detection', 'woszczyk24_interspeech', 'embeddings -score privacy preserving speaker limited-resource anonymize adresso adress identifiable'], ['Minh Nguyen, Toan Quoc Nguyen, Kishan KC, Zeyu Zhang, Thuy Vu', 'Reinforcement Learning from Answer Reranking Feedback for Retrieval-Augmented Answer Generation', 'nguyen24c_interspeech', 'odqa reward rag misaligned factual open-domain model human retrieving answering'], ['Gasser Elbanna, Zohreh Mostaani, Mathew Magimai.-Doss', 'Predicting Heart Activity from Speech using Data-driven and Knowledge-based features', 'elbanna24_interspeech', 'physiological self-supervised acoustic speaker-related underscore speech-related unexplored biological generalizability representation'], ['Jinuk Kwon, David Harwath, Debadatta Dash, Paul Ferrari, Jun Wang', 'Direct Speech Synthesis from Non-Invasive, Neuromagnetic Signals', 'kwon24_interspeech', 'brain neural meg bci neuroimaging magnetoencephalography invasive overt electrode mel-spectrograms'], ['Menglu Li, Xiao-Ping Zhang', 'Interpretable Temporal Class Activation Representation for Audio Spoofing  Detection', 'li24oa_interspeech', 'interpretability model fostering t-dcf multi-label trust decision-making transparency bonafide localize'], ['Peter Mihajlik, Yan Meng, Mate S Kadar, Julian Linke, Barbara Schuppler, Katalin Mády', 'On Disfluency and Non-lexical Sound Labeling for End-to-end Automatic Speech Recognition', 'mihajlik24_interspeech', 'label filled pause grunt conversational delete decent backchannels asr disfluent'], ['David Looney, Nikolay D. Gaubitch', 'Robust spread spectrum speech watermarking using linear prediction and deep spectral shaping', 'looney24_interspeech', 'music signal deepfake centre date attack primarily classical implication call'], ['Ashutosh Pandey, Sanha Lee, Juan Azcarreta, Daniel Wong, Buye Xu', 'All Neural Low-latency Directional Speech Extraction', 'pandey24_interspeech', 'doa quickly model embeddings capability scenario frame abrupt hand-crafted grid'], ['Benazir Mumtaz, Miriam Butt', 'Urdu Alternative Questions: A Hat Pattern', 'mumtaz24_interspeech', 'altqs spacing contour realization german accent noting prolonged featuring focus'], ['Kevin Huang, Jack Goldberg, Louis Goldstein, Shrikanth Narayanan', 'Analysis of articulatory setting for L1 and L2 English speakers using MRI data', 'huang24i_interspeech', 'acquired articulation korea posture geographical positional united india china draw'], ['Kuang Yuan, Shuo Han, Swarun Kumar, Bhiksha Raj', 'DeWinder: Single-Channel Wind Noise Reduction using Ultrasound Sensing', 'yuan24_interspeech', 'doppler ultrasonic outdoor deep-learning mitigating characteristic airflow fuse qual
11146ity treat'], ['Mohammad Amaan Sayeed, Hanan Aldarmaki', 'Spoken Word2Vec: Learning Skipgram Embeddings from Speech', 'sayeed24_interspeech', 'encode distributional semantic semantics examine similarity resulting relatedness analogous phonetic'], ['Giulia Sanguedolce, Sophie Brook, Dragos C. Gruia, Patrick A. Naylor, Fatemeh Geranmayeh', 'When Whisper Listens to Aphasia: Advancing Robust Post-Stroke Speech Recognition', 'sanguedolce24_interspeech', 'aphasiabank generalisability intervention asr ai-based wer asrs stroke standardised versatile'], ['Sidi Yaya Arnaud Yarga, Sean U N Wood', 'Neuromorphic Keyword Spotting with Pulse Density Modulation MEMS Microphones', 'yarga24_interspeech', 'spiking device gsc energy kw command bio-inspired ssc snn stage'], ['Yuzhe Wang, Anna Favaro, Thomas Thebaud, Jesus Villalba, Najim Dehak, Laureano Moro-Velazquez', 'Exploring the Complementary Nature of Speech and Eye Movements for Profiling Neurological Disorders', 'wang24ka_interspeech', 'nd disease ocular parkinson differentiated search subject thief visual visit'], ['Amrutha Prasad, Srikanth Madikeri, Driss Khalil, Petr Motlicek, Christof Schuepbach', 'Speech and Language Recognition with Low-rank Adaptation of Pretrained Models', 'prasad24_interspeech', 'xls-r finetuning whisper drop connected fully resource constraint layer training'], ['Kwangyoun Kim, Suwon Shon, Yi-Te Hsu, Prashant Sridhar, Karen Livescu, Shinji Watanabe', 'Convolution-Augmented Parameter-Efficient Fine-Tuning for Speech Recognition', 'kim24s_interspeech', 'peft bottleneck lora low-rank adapter hubert fine-tune method numerous convolution'], ['Jonathan Him Nok Lee, Mark Liberman, Martin Salzmann', 'Do we EXPECT TO find phonetic traces for syntactic traces?', 'lee24l_interspeech', 'intervening contraction theory morpho-phonological regression lenition presence posited contradict multinomial'], ['Chaofei Fan, Jaimie M. Henderson, Chris Manning, Francis R. Willett', 'Towards a Quantitative Analysis of Coarticulation with a Phoneme-to-Articulatory Model', 'fan24c_interspeech', 'ema magnitude extent resistance phoneme comprehensively across sequence replicate trigger'], ['Angelo Ortiz Tandazo, Thomas Schatz, Thomas Hueber, Emmanuel Dupoux', 'Simulating articulatory trajectories with phonological feature interpolation', 'ortiztandazo24_interspeech', 'phonology generative perception-production correlation linear co-articulation biological pearson optimisation loop'], ['Tejes Srivastava, Jiatong Shi, William Chen, Shinji Watanabe', 'EFFUSE: Efficient Self-Supervised Feature Fusion for E2E ASR in Low Resource and Multilingual Scenarios', 'srivastava24_interspeech', 'ssl model fusing ml-superb parameter performance size average superb increase'], ['David Meyer, Eitan Abecassis, Clara Fernandez-Labrador, Christopher Schroers', 'RAST: A Reference-Audio Synchronization Tool for Dubbed Content', 'meyer24b_interspeech', 'translated disengagement sync intermittent defect detection misalignment viewer siamese issue'], ['Natalia Morozova, Guanghao You, Sabine Stoll, Adrian Bangerter', 'Measuring acoustic dissimilarity of hierarchical markers in task-oriented dialogue with MFCC-based dynamic time warping', 'morozova24_interspeech', 'transition horizontal vertical acoustically block interactants unfold navigating okay vertically'], ['Daniel Escobar-Grisales, Cristian David Ríos-Urrego, Ilja Baumann, Korbinian Riedhammer, Elmar Noeth, Tobias Bocklet, Adolfo M. Garcia, Juan Rafael Orozco-Arroyave', 'It’s Time to Take Action: Acoustic Modeling of Motor Verbs to Detect Parkinson’s Disease', 'escobargrisales24_interspeech', 'yielded representation pre-trained discrimination non-motor accuracy neurocognitive model well-established frame-by-frame'], ['Abderrahim Fathan, Xiaolin Zhu, Jahangir Alam', 'On the impact of several regularization techniques on label noise robustness of self-supervised speaker verification systems', 'fathan24_interspeech', 'pls sssv investigative loss mixup clustering-based pseudo-labels thorough effect smoothing'], ['Aravind Krishnan, Badr M. Abdullah, Dietric
11146h Klakow', 'On the Encoding of Gender in Transformer-based ASR Representations', 'krishnan24_interspeech', 'layer erasure concentration prospect uncover compromising explaining hubert utilization deeper'], ['Xianrui Zheng, Guangzhi Sun, Chao Zhang, Philip C. Woodland', 'SOT Triggered Neural Clustering for Speaker Attributed ASR', 'zheng24d_interspeech', 'sdnc diarisation eval cascaded assign ami system cpwer parallel label'], ['Ariëlle Reitsema, Chenxin Li, Leanne van Lambalgen, Laura Preining, Saskia Galindo Jong, Qing Yang, Xinyi Wen, Yiya Chen', 'Perceptual Learning in Lexical Tone: Phonetic Similarity vs. Phonological Categories', 'reitsema24_interspeech', 'contour rising pitch exposure interpretation recalibrate recalibration disambiguating listener segmentally'], ['Wonjune Kang, Deb Roy', 'Prompting Large Language Models with Audio for General-Purpose Speech Summarization', 'kang24d_interspeech', 'llm speech-text text reasoning framework processing summarize input cascade summary'], ['Zehua Kcriss Li, Meiying Melissa Chen, Yi Zhong, Pinxin Liu, Zhiyao Duan', 'GTR-Voice: Articulatory Phonetics Informed Controllable Expressive Speech Synthesis', 'li24pa_interspeech', 'gtr actor professional voice framework dimension tt mastered lens tenseness'], ['Mostafa Shahin, Beena Ahmed', 'Phonological-Level Mispronunciation Detection and Diagnosis', 'shahin24_interspeech', 'mdd phoneme-level pronunciation diagnostic phoneme error false detect rate frr'], ['Tatsunari Takagi, Yukoh Wakabayashi, Atsunori Ogawa, Norihide Kitaoka', 'Text-only Domain Adaptation for CTC-based Speech Recognition through Substitution of Implicit Linguistic Information in the Search Space', 'takagi24_interspeech', 'ctc asr dra method japanese practicality english-language character-level accommodated model'], ['Rishi Jain, Bohan Yu, Peter Wu, Tejas Prabhune, Gopala Anumanchipalli', 'Multimodal Segmentation for Vocal Tract Modeling', 'jain24_interspeech', 'rt-mri articulator mri internal labeling video occluded label dataset -speaker'], ['Gowtham Premananth, Yashish M. Siriwardena, Philip Resnik, Sonia Bansal, Deanna L.Kelly, Carol Espy-Wilson', 'A Multimodal Framework for the Assessment of the Schizophrenia Spectrum', 'premananth24_interspeech', 'fusion weighted modality bi-modal unit -score gated bimodal unimodal symptom'], ['Orchid Chetia Phukan, Gautam Siddharth Kashyap, Arun Balaji Buduru, Rajesh Sharma', 'Are Paralinguistic Representations all that is needed for Speech Emotion Recognition?', 'phukan24b_interspeech', 'ptm ser sota trillsson ml-superb ptms superb fill facilitated attributed'], ['Justin Lovelace, Soham Ray, Kwangyoun Kim, Kilian Q. Weinberger, Felix Wu', 'Sample-Efficient Diffusion for Text-To-Speech Synthesis', 'lovelace24_interspeech', 'sesd less latent vall-e speech regime synthesizes auto-regressive modest impressive'], ['Tomas Arias-Vergara, Paula Andrea Pérez-Toro, Xiaofeng Liu, Fangxu Xing, Maureen Stone, Jiachen Zhuo, Jerry L. Prince, Maria Schuster, Elmar Noeth, Jonghye Woo, Andreas Maier', 'Contrastive Learning Approach for Assessment of Phonological Precision in Patients with Tongue Cancer Using MRI Data', 'ariasvergara24_interspeech', 'class clinical recognition assess high-resolution frame-wise unavailable synchronized magnetic application'], ['Manila Kodali, Sudarsana Reddy Kadiri, Paavo Alku', 'Fine-tuning of Pre-trained Models for Classification of Vocal Intensity Category from Speech Signals', 'kodali24_interspeech', 'regulate spl loud wav vec arbitrary amplitude scale label occasion'], ['Prad Kadambi, Tristan Mahr, Lucas Annear, Henry Nomeland, Julie Liss, Katherine Hustad, Visar Berisha', "How Does Alignment Error Affect Automated Pronunciation Scoring in Children's Speech?", 'kadambi24_interspeech', 'pllr deviation forced phoneme score manual computed effect magnified using'], ['Chenzi Xu, Jessica Wormald, Paul Foulkes, Philip Harrison, Vincent Hughes, Poppy Welch, Finnian Kelly, David van der Vloed', 'Voice quality in telephone speech: Comparing acoustic measures between VoIP telephone and high-quality recordings', 'xu24j_interspeech', 'cpp studio forensic condition recording harmonics-to-noise susceptible differentiation creaky breathy'], ['Yongyi Zang, Jiatong Shi, You Zhang, Ryuichi Yamamoto, Jionghao Han, Yuxun Tang, Shengyuan Xu, Wenxiao Zhao, Jing Guo, Tomoki Toda, Zhiyao Duan', 'CtrSVDD: A Benchmark Dataset and Baseline Analysis for Controlled Singing Voice Deepfake Detection', 'zang24_interspeech', 'bonafide accessible vocal publicly hour licensing method necessitate datasets controllability'], ['Xi Liu, John H.L. Hansen', 'DNN-based monaural speech enhancement using alternate analysis windows for phase and magnitude modification', 'liu24o_interspeech', 'hanning window stft enhancing phase-aware chebyshev apollo fearless quefrency evolves'], ['Dena Mujtaba, Nihar R. Mahapatra, Megan Arney, J. Scott Yaruss, Caryn Herring, Jia Bin', 'Inclusive ASR for Disfluent Speech: Cascaded Large-Scale Self-Supervised Learning with Targeted Fine-Tuning and Data Augmentation', 'mujtaba24_interspeech', 'inclusivity involuntary enriches datasets stutter curated asrs pave dataset barrier'], ['May Pik Yu Chan, Jianjing Kuang', 'Pitch-driven adjustments in tongue positions: Insights from ultrasound imaging', 'chan24_interspeech', 'pitch adjust participant range resonance sing space higher semitone production'], ['Jiatong Shi, Shih-Heng Wang, William Chen, Martijn Bartelds, Vanya Bannihatti Kumar, Jinchuan Tian, Xuankai Chang, Dan Jurafsky, Karen Livescu, Hung-yi Lee, Shinji Watanabe', 'ML-SUPERB 2.0: Benchmarking Multilingual Speech Models Across Modeling Constraints, Languages, and Datasets', 'shi24g_interspeech', 'downstream ssl setup benchmark performance find asr shallow targeted treat'], ['Khai Le-Duc, Khai-Nguyen Nguyen, Long Vo-Dang, Truong-Son Hy', 'Real-time Speech Summarization for Medical Conversations', 'leduc24_interspeech', 'summary conversation doctor-patient deployable first collaboratively standpoint posing thirdly gold'], ['Jiatong Shi, Xutai Ma, Hirofumi Inaguma, Anna Sun, Shinji Watanabe', 'MMM: Multi-Layer Multi-Residual Multi-Stream Discrete Speech Representation from Self-supervised Learning Model', 'shi24h_interspeech', 'ssl unit on-par compatibility surpass resynthesis lag various k-means codec'], ['Hazim Bukhari, Soham Deshmukh, Hira Dhamyal, Bhiksha Raj, Rita Singh', 'SELM: Enhancing Speech Emotion Recognition for Out-of-Domain Scenarios', 'bukhari24_interspeech', 'ser ood ravdess formulation situation crema-d curated inspiration performance formulate'], ['Paul Best, Santiago Cuervo, Ricard Marxer', 'Transfer Learning from Whisper for Microscopic Intelligibility Prediction', 'best24_interspeech', 'macroscopic response deep model scale predict lexical speech-in-noise word-error-rate listener'], ['Irene Smith, Morgan Sonderegger, The Spade Consortium', 'Modelled Multivariate Overlap: A method for measuring vowel merger', 'smith24_interspeech', 'distribution tension empirical modelling univariate affinity bhattacharyya extraneous unbalanced measure'], ['Tejumade Afonja, Tobi Olatunji, Sew
11146ade Ogun, Naome A. Etori, Abraham Owodunni, Moshood Yekini', 'Performant ASR Models for Medical Entities in Accented Speech', 'afonja24_interspeech', 'clinical wer drug rigorously error posing stride healthcare accelerated safety'], ['Prakash Kumar, Ye Tian, Yongwan Lim, Sophia X. Cui, Christina Hagedorn, Dani Byrd, Uttam K. Sinha, Shrikanth Narayanan, Krishna S. Nayak', 'State-of-the-art speech production MRI protocol for new 0.55 Tesla scanners', 'kumar24b_interspeech', 'rt-mri imaging platform real-time tract blurring biofeedback vocal safe slice'], ['Yiwen Shao, Shi-Xiong Zhang, Dong Yu', 'RIR-SF: Room Impulse Response Based Spatial Feature for Target Speech Recognition in Multi-Channel Multi-Speaker Scenarios', 'shao24_interspeech', 'reflection wave asr all-neural overlooking rir hinders speaker overcoming multi-talker'], ['Margaret Kroll, Kelsey Kraus', 'Optimizing the role of human evaluation in LLM-based spoken document summarization systems', 'kroll24_interspeech', 'llm paradigm replicability bertscore human-in-the-loop abstractive rouge content trustworthiness ability'], ['Robin Netzorg, Alyssa Cote, Sumi Koshin, Klo Vivienne Garoute, Gopala Krishna Anumanchipalli', 'Speech After Gender: A Trans-Feminine Perspective on Next Steps for Speech Science and Technology', 'netzorg24_interspeech', 'voice speaker texture modification vocal fail notion categorical static identity'], ['Irina-Elena Veliche, Zhuangqun Huang, Vineeth Ayyat Kochaniyan, Fuchun Peng, Ozlem Kalinli, Michael L. Seltzer', 'Towards measuring fairness in speech recognition: Fair-Speech dataset', 'veliche24_interspeech', 'demographic asr ethnicity submit geographic self-reported saying across united untranscribed'], ['Yiwen Shao, Shi-Xiong Zhang, Yong Xu, Meng Yu, Dong Yu, Daniel Povey, Sanjeev Khudanpur', 'Multi-Channel Multi-Speaker ASR Using Target Speaker’s Solo Segment', 'shao24b_interspeech', 'solo-sf array microphone circumventing alimeeting voiceprint layout discerning formidable effective'], ['Eunice Akani, Frederic Bechet, Benoît Favre, Romain Gemignani', 'Unified Framework for Spoken Language Understanding and Summarization in Task-Based Human Dialog processing', 'akani24_interspeech', 'summary dialogue task-related decoda goal-oriented summarizing information concise multitask process'], ['Sewade Ogun, Abraham T. Owodunni, Tobi Olatunji, Eniola Alese, Babatunde Oladimeji, Tejumade Afonja, Kayode Olaleye, Naome A. Etori, Tosin Adewumi', '1000 African Voices: Advancing inclusive multi-speaker multi-accent speech synthesis', 'ogun24_interspeech', 'persona creation automated geography data-rich continent under-represented content accent like'], ['Minxue Niu, Mimansa Jaiswal, Emily Mower Provost', 'From Text to Emotion: Unveiling the Emotion Annotation Capabilities of LLMs', 'niu24d_interspeech', 'gpt- human model impact underestimate training potential assisting automating underscore'], ['Dawei Liang, Alice Zhang, David Harwath, Edison Thomaz', 'Improving Audio Classification with Low-Sampled Microphone Input: An Empirical Study Using Model Self-Distillation', 'liang24_interspeech', 'khz panns sampling mobile high-quality traction lower rate low-quality wearable'], ['Emily P. Ahn, Eleanor Chodroff, Myriam Lapierre, Gina-Anne Levow', 'The Use of Phone Categories and Cross-Language Modeling for Phone Alignment of Panãra', 'ahn24d_interspeech', 'forced granularity model pretrained language-specific low-resource acoustic language amazonian broadening'], ['Ioana Colgiu, Laura Spinu, Rajiv Rao, Yasaman Rafat', 'Bilingual Rhotic Production Patterns: A Generational Comparison of Spanish-English Bilingual Speakers in Canada', 'colgiu24_interspeech', 'cli trill late early drift approximant tap alveolar phonemic producing'], ['Nana Lin, Youxiang Zhu, Xiaohui Liang, John A. Batsis, Caroline Summerour', 'Analyzing Multimodal Features of Spontaneous Voice Assistant Commands for Mild Cognitive Impairment Detection', 'lin24l_interspeech', 'mci intent task fusion in-home subdomains read progressing dementia classification'], ['K R Prajwal, Triantafyllos Afouras, Andrew Zisserman', 'Speech Recognition Models are Strong Lip-readers', 'prajwal24_interspeech', 'lip-reading pre-trained asr lip video mapping perform model lr regime'], ['Yuxun Tang, Yuning Wu, Jiatong Shi, Qin Jin', 'SingOMD: Singing Oriented Multi-resolution Discrete Representation Construction from Speech Models', 'tang24c_interspeech', 'ssl generation discretizing discretized resampling necessitates wherein feature resynthesis encounter'], ['Vrushank Changawala, Frank Rudzicz', 'Whister: Using Whisper’s representations for Stuttering detection', 'changawala24_interspeech', 'split data dysfluency cross-corpora specifically generalizable surpassing segment -second frozen'], ['Krishna C. Puvvada, Piotr Żelasko, He Huang, Oleksii Hrinchuk, Nithin Rao Koluguri, Kunal Dhawan, Somshubra Majumdar, Elena Rastorgueva, Zhehuai Chen, Vitaly L
11146avrukhin, Jagadeesh Balam, Boris Ginsburg', 'Less is More: Accurate Speech Recognition &amp; Translation without Web-Scale Data', 'puvvada24_interspeech', 'model canary owsm training dynamic blending open-sourced state-of-the noise-robust whisper'], ['Junming Yuan, Ying Shi, LanTian Li, Dong Wang, Askar Hamdulla', 'Few-Shot Keyword Spotting from Mixed Speech', 'yuan24b_interspeech', 'pre-training fine-tuning solve keywords phase blended ssl-based effective struggle hubert'], ['Xintong Wang, Mingqian Shi, Ye Wang', 'Pitch-Aware RNN-T for Mandarin Chinese Mispronunciation Detection and Diagnosis', 'wang24la_interspeech', 'mdd stateless stage model pitch hubert scarcity acceptance two-stage solely'], ['Annika Heuser, Tyler Kendall, Miguel del Rio, Quinn McNamara, Nishchal Bhandari, Corey Miller, Migüel Jetté', 'Quantification of stylistic differences in human- and ASR-produced transcripts of African American English', 'heuser24_interspeech', 'aae verbatim transcriber asr human compounded underrepresented speech assess morphosyntactic'], ['James Tavernor, Yara El-Tawil, Emily Mower Provost', 'The Whole Is Bigger Than the Sum of Its Parts: Modeling Individual Annotators to Capture Emotional Variability', 'tavernor24_interspeech', 'emotion distribution label omits inter-annotator nuanced lose cross-corpus address nuance'], ['Pavan Kalyan, Preeti Rao, Preethi Jyothi, Pushpak Bhattacharyya', 'Emotion Arithmetic: Emotional Speech Synthesis via Weight Space Interpolation', 'kalyan24_interspeech', 'vector behaviour task idea negation generation steer subtracting steering model'], ['Rui Liu, Jiatian Xi, Ziyue Jiang, Haizhou Li', 'FluentEditor: Text-based Speech Editing by Considering Acoustic and Prosody Consistency', 'liu24p_interspeech', 'fluency tse edited region -lab consistent constraint original vctk segment'], ['Yiru Zhang, Linyu Yao, Qun Yang', 'OR-TSE: An Overlap-Robust Speaker Encoder for Target Speech Extraction', 'zhang24p_interspeech', 'tse reference overlap embeddings pre-enrolled disregarding mixture non-overlapping attentive mainstream'], ['Hyung Yong Kim, Byeong-Yeol Kim, Yunkyu Lim, Jihwan Park, Shukjae Choi, Yooncheol Ju, Jinseok Park, Youshin Lim, Seung Woo Yu, Hanbin Lee, Shinji Watanabe', 'Self-training ASR Guided by Unsupervised ASR Teacher', 'kim24t_interspeech', 'pseudo-target vec dataset labeled uasr pseudo-targets asr-related layer test-other test-clean'], ['Suyuan Liu, Molly Babel, Jian Zhu', 'A comparison of voice similarity through acoustics, human perception and deep neural network (DNN) speaker verification systems', 'liu24q_interspeech', 'listener perceptual judgment assessed acoustic made dis generated happens score'], ['Zuheyra Tokac, Jennifer Cole', 'Phonological Symmetry Does Not Predict Generalization of Perceptual Adaptation to Vowels', 'tokac24_interspeech', 'lowered inventory variant -vowel identification lasting boundary symmetrical assess biasing'], ['Jingyi Feng, Yusuke Yasuda, Tomoki Toda', 'Exploring the Robustness of Text-to-Speech Synthesis Based on Diffusion Probabilistic Models to Heavily Noisy Transcriptions', 'feng24d_interspeech', 'tt volume transcription data flow-based diffusion-based mitigating suffers autoregressive extremely'], ['Annika Heuser, Jianjing Kuang', 'Information-theoretic hypothesis generation of relative cue weighting for the voicing contrast', 'heuser24b_interspeech', 'sae context child gleaned perceptual different obstruent coda helpful validated'], ['Anton de la Fuente, Dan Jurafsky', 'A layer-wise analysis of Mandarin and English suprasegmentals in SSL speech models', 'delafuente24_interspeech', 'wav vec representation layer suprasegmental later contextual stress represent mainly'], ['Linhan Ma, Dake Guo, Kun Song, Yuepeng Jiang, Shuai Wang, Liumeng Xue, Weiming Xu, Huan Zhao, Binbin Zhang, Lei Xie', 'WenetSpeech4TTS: A 12,800-hour Mandarin TTS Corpus for Large Speech Generation Model Benchmark', 'ma24d_interspeech', 'segment subset huggingface text-to-speech open-sourced audio-text multi-domain fair data eliminating'], ['Sri Harsha Dumpala, Katerina Dikaios, Abraham Nunes, Frank Rudzicz, Rudolf Uher, Sageev Oore', 'Self-Supervised Embeddings for Detecting Individual Symptoms of Depression', 'dumpala24b_interspeech', 'ssl severity predicting identifying small-sized depressive elucidate speech impacting learning'], ['Nan Chen, Yonghe Wang, Feilong Bao', 'Knowledge-Preserving Pluggable Modules for Multilingual Speech Translation Tasks', 'chen24w_interspeech', 'existing resampling retraining language regularization new performance cost model increase'], ['Oguzhan Baser, Kaan Kale, Sandeep P. Chinchali', 'SecureSpectra: Safeguarding Digital Identity from Deep Fake Threats via Intelligent Signatures', 'baser24_interspeech', 'misinformation irreversible unauthorized strike mozilla datasets delicate deepfake defense authentication'], ['Xiaohan Shi, Xingfeng Li, Tomoki Toda', 'Multimodal Fusion of Music Theory-Inspired and Self-Supervised Representations for Improved Emotion Recognition', 'shi24i_interspeech', 'mer modality modality-specific deepen effectiveness comprehensively hinder challenge underscore handcrafted'], ['Max Morrison, Cameron Churchwell, Nathan Pruyne, Bryan Pardo', 'Fine-Grained and Interpretable Neural Speech Editing', 'morrison24_interspeech', 'disentangled volume attribute representation identity dialogue imperfection timbral pronunciation fixing'], ['Matthew McNeill, Rivka Levitan', 'Autoregressive cross-interlocutor attention scores meaningfully capture conversational dynamics', 'mcneill24_interspeech', 'dialogue entrainment organize historical human-human analyzes human-machine partner qual
11146itative history'], ['Yuning Wu, Chunlei Zhang, Jiatong Shi, Yuxun Tang, Shan Yang, Qin Jin', 'TokSing: Singing Voice Synthesis based on Discrete Tokens', 'wu24q_interspeech', 'melody svs token mel intermediate spectrogram offer selfsupervised witness discretization'], ['Viyazonuo Terhiija, Priyankoo Sarmah', 'Voiced and voiceless laterals in Angami', 'terhiija24_interspeech', 'distinction voicing delf cross-linguistically aspirated hnr exception context rare trait'], ['Linhan Ma, Xinfa Zhu, Yuanjun Lv, Zhichao Wang, Ziqian Wang, Wendi He, Hongbin Zhou, Lei Xie', 'Vec-Tok-VC+: Residual-enhanced Robust Zero-shot Voice Conversion with Progressive Constraints in a Dual-mode Training Strategy', 'ma24e_interspeech', 'training-inference content process multi-codebook mismatch prompt-based semantic similarity loss decoupling'], ['Jehyun Kyung, Serin Heo, Joon-Hyuk Chang', 'Enhancing Multimodal Emotion Recognition through ASR Error Compensation and LLM Fine-Tuning', 'kyung24_interspeech', 'mer asr-generated inaccuracy text cmt counteract blend system compromised nuance'], ['Slava Shechtman, Avihu Dekel', 'Low Bitrate High-Quality RVQGAN-based Discrete Speech Tokenizer', 'shechtman24_interspeech', 'tokenizers audio open-source reconstruction publicly token pcm speech-only tokenization bps'], ['Junghun Kim, Ka Hyun Park, Hoyoung Yoon, U Kang', 'Domain-Aware Data Selection for Speech Classification via Meta-Reweighting', 'kim24u_interspeech', 'domain instance source utilizing disorder target accurate softly specific given'], ['Jiali Cheng, Mohamed Elgaar, Nidhi Vakil, Hadi Amiri', 'CogniVoice: Multimodal and Multilingual Fusion Networks for Mild Cognitive Impairment Assessment from Spontaneous Speech', 'cheng24c_interspeech', 'mci mmse taukadial mini-mental point shortcut mitigates reliance decline rmse'], ['Shaowen Chen, Tomoki Toda', 'QHM-GAN: Neural Vocoder based on Quasi-Harmonic Modeling', 'chen24x_interspeech', 'qhm consumption vocoders speech quality network signal hifi-gan prominently black-box'], ['Tian-Hao Zhang, Xinyuan Qian, Feng Chen, Xu-Cheng Yin', 'Transmitted and Aggregated Self-Attention for Automatic Speech Recognition', 'zhang24q_interspeech', 'map attention layer transformer respectively information outstanding aishell- previous cer'], ['Bohan Li, Feiyu Shen, Yiwei Guo, Shuai Wang, Xie Chen, Kai Yu', 'On the Effectiveness of Acoustic BPE in Decoder-Only TTS', 'li24qa_interspeech', 'token slm speech setting semantic discretizing byte-pair libritts shorten favorable'], ['Tahir Javed, Janki Nawale, Sakshi Joshi, Eldho George, Kaushal Bhogale, Deovrat Mehendale, Mitesh M. Khapra', 'LAHAJA: A Robust Multi-accent Benchmark for Evaluating Hindi ASR Systems', 'javed24_interspeech', 'india diverse extempore north-east district accent sourced find existing terminology'], ['Alexis DeMaere, Nicole van Rootselaar, Fangfang Li, Robbin Gibb, Claudia L. R. Gonzalez', 'On the relationship between speech production and vocabulary size in 3-5 year olds', 'demaere24_interspeech', 'six-month multifaceted preschool complimentary null receptive versa vice sex comprehension'], ['Jens Heitkaemper, Joe Caroselli, Arun Narayanan, Nathan Howard', 'TfCleanformer: A streaming, array-agnostic, full- and sub-band modeling front-end for robust ASR', 'heitkaemper24_interspeech', 'enhancement upon non-causal agnostic multiple publication ablation multi-channel outperforming array'], ['Jaeuk Lee, Sohee Jang, Joon-Hyuk Chang', 'Neural ATSM: Fully Neural Network-based Adaptive Time-Scale Modification Using Sentence-Specific Dynamic Control', 'lee24m_interspeech', 'mfa phoneme sentence rate speaking networks-based scale tailoring phoneme-specific montreal'], ['Sara Ng, Gina-Anne Levow, Mari Ostendorf, Richard Wright', 'Investigating the Influence of Stance-Taking on Conversational Timing of Task-Oriented Speech', 'ng24b_interspeech', 'behavior stance turn-taking conversation northwest measurably pacific negotiation known speaker'], ['Darshan Prabhu, Yifan Peng, Preethi Jyothi, Shinji Watanabe', 'MULTI-CONVFORMER: Extending Conformer with Multiple Convolution Kernels', 'prabhu24_interspeech', 'module variant reexamined modelling e-branchformer local conformers rival eff
11146icient asr'], ['Yuan Gao, Hao Shi, Chenhui Chu, Tatsuya Kawahara', 'Speech Emotion Recognition with Multi-level Acoustic and Semantic Information Extraction and Interaction', 'gao24f_interspeech', 'ser extractor emotional asr system embeddings learn module existing feature'], ['Junwen Duan, Fangyuan Wei, Hong-Dong Li, Jin Liu', 'Pre-trained Feature Fusion and Matching for Mild Cognitive Impairment Detection', 'duan24_interspeech', 'mci language-agnostic diagnosis challenge taukadial delaying diagnose expressivity chinese-english progression'], ['Takuhiro Kaneko, Hirokazu Kameoka, Kou Tanaka, Yuto Kondo', 'FastVoiceGrad: One-step Diffusion-Based Voice Conversion with Adversarial Conditional Diffusion Distillation', 'kaneko24_interspeech', 'multi-step inference reconsidering any-to-any one-shot dozen reverse performance high attracted'], ['Sangwon Ryu, Heejin Do, Yunsu Kim, Gary Geunbae Lee, Jungseul Ok', 'Key-Element-Informed sLLM Tuning for Document Summarization', 'ryu24_interspeech', 'llm proprietary key relevance high-quality element instructs fee hallucination low'], ['Haechan Kim, Junho Myung, Seoyoung Kim, Sungpah Lee, Dongyeop Kang, Juho Kim', 'LearnerVoice: A Dataset of Non-Native English Learners’ Spontaneous Speech', 'kim24v_interspeech', 'ungrammatical vanilla disfluency consisting expression learner self-repairs datasets transcription attributable'], ['Kalvin Chang, Yi-Hui Chou, Jiatong Shi, Hsuan-Ming Chen, Nicole Holliday, Odette Scharenborg, David R. Mortensen', 'Self-supervised Speech Representations Still Struggle with African American Vernacular English', 'chang24d_interspeech', 'aave ssl variety mae asr gap model welldocumented marginalized xls-r'], ['Kaushal Santosh Bhogale, Deovrat Mehendale, Niharika Parasa, Sathish Kumar Reddy G, Tahir Javed, Pratyush Kumar, Mitesh M. Khapra', 'Empowering Low-Resource Language ASR via Large-Scale Pseudo Labeling', 'bhogale24_interspeech', 'pseudo-labeling benchmark youtube multiple labeled pseudo-labeled data existing evaluator augmenting'], ['Sai Harshitha Aluru, Jhansi Mallela, Chiranjeevi Yarra', 'Post-Net: A linguistically inspired sequence-dependent transformed neural architecture for automatic syllable stress detection', 'aluru24_interspeech', 'sota dependency supervised unsupervised isle existing model word syllable-level time-delay'], ['Conor Atkins, Ian Wood, Mohamed Ali Kaafar, Hassan Asghar, Nardine Basta, Michal Kepkowski', 'ConvoCache: Smart Re-Use of Chatbot Responses', 'atkins24_interspeech', 'prefetching coherence find prompt latency generative reuses chatbots caching reduce'], ['Liisa Rätsep, Rasmus Lellep, Mark Fishel', 'Enabling Conversational Speech Synthesis using Noisy Spontaneous Data', 'ratsep24_interspeech', 'lack stylistically read style text-to-speech datasets sample multi-style estonian compromise'], ['Jhansi Mallela, Sai Harshitha Aluru, Chiranjeevi Yarra', 'A comparative analysis of sequential models that integrate syllable dependency for automatic syllable stress detection', 'mallela24_interspeech', 'grus non-sequential stress-related overlook isle operated sequence lstms rnns identifies'], ['Yuanjun Lv, Hai Li, Ying Yan, Junhui Liu, Danming Xie, Lei Xie', 'FreeV: Free Lunch For Vocoders Through Pseudo Inversed Mel Filter', 'lv24_interspeech', 'apnet initialization inference speed amplitude parameter pursuing checkpoint streamlined mitigates'], ['Jiayan Lin, Shenghui Lu, Hukai Huang, Wenhao Guan, Binbin Xu, Hui Bu, Qingyang Hong, Lin Li', 'MinSpeech: A Corpus of Southern Min Dialect for Automatic Speech Recognition', 'lin24m_interspeech', 'hour hokkien diversely sourced encompassing audio hubert kaldi conformer cultural'], ['Chetan Sharma, Vaishnavi Chandwanshi, Prasanta Kumar Ghosh', 'A comparative study of the impact of voiceless alveolar and palato-alveolar sibilants in English on lip aperture and protrusion during VCV production', 'sharma24_interspeech', 'sibilant case usc change rounding higher vowel classification displacement irrespective'], ['Srija Anand, Praveen Srinivasa Varadhan, Ashwin Sankar, Giri Raju, Mitesh M. Khapra', 'Enhancing Out-of-Vocabulary Performance of Indian TTS Systems for Practical Applications through Low-Effort Data Strategies', 'anand24_interspeech', 'oov tamil hindi word benchmark containing code-mixing economically vocabulary artist'], ['Changli Tang, Wenyi Yu, Guangzhi Sun, Xianzhao Chen, Tian Tan, Wei Li, Jun Zhang, Lu Lu, Zejun Ma, Yuxuan Wang, Chao Zhang', 'Can Large Language Models Understand Spatial Audio?', 'tang24d_interspeech', 'llm fsr lse ssl amidst inferential llm-based via paving mae'], ['Praveen Srinivasa Varadhan, Ashwin Sankar, Giri Raju, Mitesh M Khapra', 'Rasa: Building Expressive Speech Synthesis Systems for Indian Languages in Low-resource Settings', 'srinivasavaradhan24_interspeech', 'neutral hour emotion data expressiveness prioritizing ekman assamese mushra resource-constrained'], ['Haotian Tan, Sakriani Sakti', 'Contrastive Feedback Mechanism for Simultaneous Speech Translation', 'tan24b_interspeech', 'sst cfm policy unstable prediction decision delaying overlook must-c undesired'], ['Marvin Tammen, Tsubasa Ochiai, Marc Delcroix, Tomohiro N
11146akatani, Shoko Araki, Simon Doclo', 'Array Geometry-Robust Attention-Based Neural Beamformer for Moving Speakers', 'tammen24_interspeech', 'asa module tuning manual transform-average-concatenate microphone aggregator channel necessitating mask-based'], ['Yi Lu, Yuankun Xie, Ruibo Fu, Zhengqi Wen, Jianhua Tao, Zhiyong Wang, Xin Qi, Xuefei Liu, Yongwei Li, Yukun Liu, Xiaopeng Wang, Shuchen Shi', 'Codecfake: An Initial Dataset for Detecting LLM-based Deepfake Audio', 'lu24f_interspeech', 'generation vocoder add codec neural process waveform multi-step skipping final'], ['Darshan Prabhu, Abhishek Gupta, Omkar Nitsure, Preethi Jyothi, Sriram Ganapathy', 'Improving Self-supervised Pre-training using Accent-Specific Codebooks', 'prabhu24b_interspeech', 'accent asr mozilla seldom invariance finetuning learnable learning trainable refined'], ['Mohamed Osman, Daniel Z. Kaplan, Tamer Nadeem', 'SER Evals: In-domain and Out-of-domain benchmarking for speech emotion recognition', 'osman24_interspeech', 'benchmark ssl model generalization diverse logit stride assess generalizable advent'], ['Socrates Vakirtzian, Chara Tsoukala, Stavros Bompolas, Katerina Mouzou, Vivian Stamou, Georgios Paraskevopoulos, Antonios Dimakis, Stella Markantonatou, Angela Ralli, Antonios Anastasopoulos', 'Speech Recognition for Greek Dialects: A Challenging Benchmark', 'vakirtzian24_interspeech', 'variety language overlooked encompassing non-standard asr convention arising impressive cross-lingual'], ['Vivian G. Li', 'In search of structure and correspondence in intra-speaker trial-to-trial variability', 'li24ra_interspeech', 'repetition point measurement position regulated distribution regulation study actively distributional'], ['Martino Ciaperoni, Athanasios Katsamanis, Aristides Gionis, Panagiotis Karras', 'Beam-search SIEVE for low-memory speech recognition', 'ciaperoni24_interspeech', 'beam memory overhead eliminates runtime search via linearly bottleneck viterbi'], ['Joonas Kalda, Tanel Alumae, Martin Lebourdais, Hervé Bredin, Séverin Baroudi, Ricard Marxer', 'TalTech-IRIT-LIS Speaker and Language Diarization Systems for DISPLACE 2024', 'kalda24_interspeech', 'track ensemble team submission challenge separation embedding vbx pyannote ahc'], ['Zhenxiong Tan, Xinyin Ma, Gongfan Fang, Xinchao Wang', 'LiteFocus: Accelerated Diffusion Inference for Long Audio Synthesis', 'tan24c_interspeech', 'clip -second attention latent longer tta complicates diffusion-based model designated'], ['Tianhao Wang, Lantian Li, Dong Wang', 'SE/BN Adapter: Parametric Efficient Domain Adaptation for Speaker Recognition', 'wang24ma_interspeech', 'fine-tuning pre-trained well-optimized inefficiency competes cn-celeb freezing ample squeeze-and-excitation inspiration'], ['Zhenyu Zhou, Shibiao Xu, Shi Yin, Lantian Li, Dong Wang', 'A Comprehensive Investigation on Speaker Augmentation for Speaker Recognition', 'zhou24f_interspeech', 'vtlp perturbation potential hinting delve cn-celeb new pivotal underscore proficient'], ['Vrunda N. Sukhadia, Shammur Absar Chowdhury', 'Children’s Speech Recognition through Discrete Token Enhancement', 'sukhadia24_interspeech', 'privacy single-view data multi-view asr degrading transforming scarcity low-resource approximate'], ['Dehua Tao, Tan Lee, Harold Chui, Sarah Luk', 'Learning Representation of Therapist Empathy in Counseling Conversation Using Siamese Hierarchical Attention Network', 'tao24b_interspeech', 'embeddings contrastive rating encoder turn learn loss two-level positively subjectively'], ['Marianne de Heer Kloots, Willem Zuidema', 'Human-like Linguistic Biases in Neural Speech Models: Phonetic Categorization and Phonotactic Constraints in Wav2Vec2.0', 'deheerkloots24_interspeech', 'bias controlled stimulus phonotactically individual amplified sound unit localize embed'], ['Ajinkya Kulkarni, Atharva Kulkarni, Miguel Couceiro, Isabel Trancoso', 'Unveiling Biases while Embracing Sustainability: Assessing the Dual Challenges of Automatic Speech Recognition Systems', 'kulkarni24_interspeech', 'asr bias carbon mm massively sota consumption emission offering whisper'], ['Franziska Braun, Sebastian P. Bayerl, Florian Hönig, Hartmut Lehfeld, Thomas Hillemacher, Tobias Bocklet, Korbinian Riedhammer', 'Infusing Acoustic Pause Context into Text-Based Dementia Assessment', 'braun24_interspeech', 'cognitive impairment test biomarker exclusion alzheimer cross-attention mild non-invasive alongside'], ['Xiaolou Li, Zehua Liu, Chen Chen, Lantian Li, Li Guo, Dong Wang', 'Zero-Shot Fake Video Detection by Audio-Visual Consistency', 'li24ta_interspeech', 'genuine audio the-art delineated state-of vsr content one-class anchored advocated'], ['Heejin Do, Wonjun Lee, Gary Geunbae Lee', 'Acoustic Feature Mixup for Balanced Multi-aspect Pronunciation Assessment', 'do24_interspeech', 'tailor non-linearly speechocean error-rate hint interpolating suit enriched imbalance mispronunciation'], ['Ashish Mittal, Darshan Prabhu, Sunita Sarawagi, Preethi Jyothi', 'SALSA: Speedy ASR-LLM Synchronous Aggregation', 'mittal24_interspeech', 'llm decoder asr coupling low-resource mismatch tokenizers layer harnessing fleurs'], ['Takuma Okamoto, Yamato Ohtani, Sota Shimizu, Tomoki Toda, Hisashi Kawai', 'Challenge of Singing Voice Synthesis Using Only Text-To-Speech Corpus With FIRNet Source-Filter Neural Vocoder', 'okamoto24_interspeech', 'svs tt phoneme shift duration input prototyped hifi-gan acoustic lyric'], ['Yuepeng Jiang, Tao Li, Fengyu Yang, Lei Xie, Meng Meng, Yujun Wang', 'Towards Expressive Zero-Shot Speech Synthesis with Hierarchical Prosody Modeling', 'jiang24d_interspeech', 'timbre expressiveness global model naturalness synthesized adaptor introduce diffusion hierarchically'], ['Chen Chen, Zehua Liu, Xiaolou Li, Lantian Li, Dong Wang', 'CNVSRC 2023: The First Chinese Continuous Visual Speech Recognition Challenge', 'chen24y_interspeech', 'vsr single-speaker 
11146cnceleb summarises encompassing comprehensively task registered org probe'], ['Joyshree Chakraborty, Leena Dihingia, Priyankoo Sarmah, Rohit Sinha', 'On Comparing Time- and Frequency-Domain Rhythm Measures in Classifying Assamese Dialects', 'chakraborty24_interspeech', 'quot assam district amplitude-modulated wind variety sun quadratic domain comprise'], ['Rui Liu, Zening Ma', 'Emotion-Aware Speech Self-Supervised Representation Learning with Intensity Knowledge', 'liu24r_interspeech', 'emotion masking speech-emotion -lab prior npc overlook emotion-related neglecting prevailing'], ['Vasista Sai Lodagala, Abhishek Biswas, Shoutrik Das, Jordan F, S Umesh', 'All Ears: Building Self-Supervised Learning based ASR models for Indian Languages at scale', 'lodagala24_interspeech', 'ssl downstream benchmark speech signifies out-perform curate abundance superb pre-train'], ['Yo-Han Park, Wencke Liermann, Yong-Seok Choi, Seung Hi Kim, Jeong-Uk Bang, Seung Yun, Kong Joo Lee', 'Backchannel prediction, based on who, when and what', 'park24b_interspeech', 'talk counseling backchanneling conversation model information backchannels interpersonal piece amp'], ['Haoqin Sun, Shiwan Zhao, Xiangyu Kong, Xuechen Wang, Hui Wang, Jiaming Zhou, Yong Qin', 'Iterative Prototype Refinement for Ambiguous Speech Emotion Recognition', 'sun24e_interspeech', 'ipr label ser precise ambiguity subtlety daunting reinforcing urgent proving'], ['Bao Thang Ta, Minh Tu Le, Van Hai Do, Huynh Thi Thanh Binh', 'Enhancing No-Reference Speech Quality Assessment with Pairwise, Triplet Ranking Losses, and ASR Pretraining', 'ta24_interspeech', 'sqa mse sample distinction relative nisqa symmetrically enforcing garnered among'], ['Kumar Neelabh, Vishnu Sreekumar', 'From Sound to Meaning in the Auditory Cortex: A Neuronal Representation and Classification Analysis', 'neelabh24_interspeech', 'semantic category understood informative neural acoustic repertoire vocalization incoming listened'], ['Roland Hartanto, Sakriani Sakti, Koichi Shinoda', 'MSDET: Multitask Speaker Separation and Direction-of-Arrival Estimation Training', 'hartanto24_interspeech', 'location sms-wsj location-based estoi si-sdr point azimuth doa permutation angle'], ['Bao Thang Ta, Van Hai Do, Huynh Thi Thanh Binh', 'Enhancing Non-Matching Reference Speech Quality Assessment through Dynamic Weight Adaptation', 'ta24b_interspeech', 'fixed rigidly pristine roll nisqa sqa training assigns multitask accommodate'], ['Bonian Jia, Huiyao Chen, Yueheng Sun, Meishan Zhang, Min Zhang', 'LLM-Driven Multimodal Opinion Expression Identification', 'jia24_interspeech', 'oei dataset text subtlety underlining encompass delivering depression mirror advancing'], ['Takayuki Arai, Ryohei Suzuki, Chandler Earp, Shinya Tsuji, Keiko Ochi', 'Production of phrases by mechanical models of the human vocal tract', 'arai24_interspeech', 'cam model umeda rotating movable successfully pipe japanese three-tube morning'], ['Vishal Gourav, Ankit Tyagi, Phanindra Mankale', 'Faster Vocoder: a multi threading approach to achieve low latency during TTS Inference', 'gourav24_interspeech', 'cpl customer time fast text get service processing buy ssml'], ['Aanchan Mohan, Monideep Chakraborti, Katelyn Eng, Nailia Kushaeva, Mirjana Prpa, Jordan Lewis, Tianyi Zhang, Vince Geisler, Carol Geisler', 'A powerful and modern AAC composition tool for impaired speakers', 'mohan24_interspeech', 'message contextually software empowering large-language relevant communication augmentative authenticity able'], ['Grzegorz P. Mika, Konrad Zieli´nski, Paweł Cyrta, Marek Grzelec', 'VoxFlow AI: wearable voice converter for atypical speech', 'mika24_interspeech', 'interaction loudspeaker neurological tell amp demonstration real-life daily usability lie'], ['Sai Akarsh, Vamshiraghusimha Narasinga, Anil Kumar Vuppala', 'Stress transfer in speech-to-speech machine translation', 'akarsh24_interspeech', 'file speech inclusivity diminishing content sector hindering give monotonous reproducing'], ['Takuma Okamoto, Yamato Ohtani, Hisashi Kawai', 'Mobile PresenTra: NICT fast neural text-to-speech system on smartphones with incremental inference of MS-FC-HiFi-GAN for law-latency synthesis', 'okamoto24b_interspeech', 'vocoder smartphone high-fidelity transformer encoder decoder prototyped convnext attendee low-latency'], ['Alex Peiró-Lilja, José Giraldo, Martí Llopart-Font, Carme Armentano-Oller, Baybars Külebi, Mireia Farrús', 'Multi-speaker and multi-dialectal Catalan TTS models for video gaming', 'peirolilja24_interspeech', 'multi-accent demo game exported export unity reply architecture reproduced execution'], ['Juliana Francis, Éva Székely, Joakim Gustafson', 'ConnecTone: a modular AAC system prototype with contextual generative text prediction and style-adaptive conversational TTS', 'francis24_interspeech', 'implement testing transformative adjustable augmentative anticipate technology language context-sensitive delivery'], ['Mahdin Rohmatillah, Bryan Gautama Ngo, Willianto Sulaiman, Po-Chuan Chen, Jen-Tzung Chien', 'Reliable dialogue system for facilitating student-counselor communication', 'rohmatillah24_interspeech', 'counselor student mental waiting health history period university dashboard in-person'], ['Yashwardhan Chaudhuri, Paridhi Mundra, Arnesh Batra, Orchid Chetia Phukan, Arun Balaji Buduru', 'ASGIR: audio spectrogram transformer guided classification and information retrieval for birds', 'chaudhuri24_interspeech', 'bird ecological sound habitat conservation wikipedia pivotal localize geographical judging'], ['Devyani Koshal, Orchid Chetia Phukan, Sarthak Jain, Arun Balaji Buduru, Rajesh Sharma', 'PERSONA: an application for emotion recognition, gender recognition and age estimation', 'koshal24_interspeech', 'ptm task obviates model stride concurrently deploying less comparatively learning'], ['Kesavaraj V, Charan Devarkonda, Vamshiraghusimha Narasinga, Anil Kumar Vuppala', 'Custom wake word detection', 'v24_interspeech', 'keyword open-vocabulary audio-text text embedding personalizing suffered preventing knowledge audio'], ['Song Chen, Mandar Gogate, Kia Dashtipour, Jasper Kirton-Wingate, Adeel Hussain, Faiyaz Doctor, Tughrul Arslan, Amir Hussain', 'Edged based audio-visual speech enhancement demonstrator', 'chen24z_interspeech', 'hearing aid customizing phone noisy anticipate assistive smartphone healthcare advancing'], ['Arif Reza Anway, Bryony Buck, Mandar Gogate, Kia Dashtipour, Michael Akeroyd, Amir Hussain', 'Real-Time Gaze-directed speech enhancement for audio-visual hearing-aids', 'anway24_interspeech', 'avse gaze eye nose estimation angle pose enhancing head hearing'], ['Abhishek Kumar, Srikanth Konjeti, Jithendra Vepa', 'Detection of background agents speech in contact centers', 'kumar24c_interspeech', 'conversation call inadvertently unintended utilise nearby mitigating proximity security quality'], ['Sarthak Jain, Orchid Chetia Phukan, Arun Balaji Buduru, Rajesh Sharma', 'The reasonable effectiveness of speaker embeddings for violence detection', 'jain24b_interspeech', 'avd ssl model sota ptms recognition harm preventing hinder environment'], ['Giovanni Morrone, Enrico Zovato, Fabio Brugnara, Enrico Sartori, Leonardo Badino', 'A toolkit for joint speaker diarization and identification with application to speaker-attributed ASR', 'morrone24_interspeech', 'configuration use-case institutional analytics speaker-related multiple registered user-friendly web-based modular'], ['Leonie Schade, Nico Dallmann, Olcay Tük, Stefan Lazarov, Petra Wagner', 'Understanding “understanding”: presenting a richly annotated multimodal corpus of dyadic interaction', 'schade24_interspeech', 'annotation non-verbal explanation behaviour past modality gaze explaining several level'], ['Joao Vitor Possamai de Menezes, Arne-Lukas Fietkau, Tom Diener, Steffen Kurb
11146is, Peter Birkholz', 'A demonstrator for articulation-based command word recognition', 'possamaidemenezes24_interspeech', 'software classification device record recording measuring operating articulation system single'], ['Nigel G. Ward, Andres Segura', 'Pragmatically similar utterance finder demonstration', 'ward24b_interspeech', 'similarity participant retrieves listens letting viewer model prospect hear identifies'], ['Elena Ryumina, Dmitry Ryumin, Alexey Karpov', 'OCEAN-AI: open multimodal framework for personality traits assessment and HR-processes automatization', 'ryumina24_interspeech', 'pta module behaving responsibility siamese thinking feeling including automate human'], ['Paridhi Mundra, Manik Sharma, Yashwardhan Chaudhuri, Orchid Chetia Phukan, Arun Balaji Buduru', 'VoxMed: one-step respiratory disease classifier using digital stethoscope sounds', 'mundra24_interspeech', 'patient github classify icbhi ast portugal greece illness recording repository'], ['Sarthak Sharma, Orchid Chetia Phukan, Drishti Singh, Arun Balaji Buduru, Rajesh Sharma', 'AVR: synergizing foundation models for audio-visual humor detection', 'sharma24b_interspeech', 'textual reliance asr circumvents hinge lean necessitating centered intricate eliminating'], ['Harm Lameris, Joakim Gustafson, Éva Székely', 'CreakVC: a voice conversion tool for modulating creaky voice', 'lameris24_interspeech', 'one-shot creak phonation representation level plotting human-in-the-loop cue finetuned modulate'], ['Yu-Sheng Tsao, Yung-Chang Hsu, Jiun-Ting Li, Siang-Hong Weng, Tien-Hong Lo, Berlin Chen', 'EZTalking: English assessment platform for teachers and students', 'tsao24_interspeech', 'feedback capt learning exercise pronunciation streamlines portfolio mock ai-powered empowers'], ['Bramhendra Koilakuntla, Prajesh Rana, Paras Ahuja, Srikanth Konjeti, Jithendra Vepa', 'Leveraging large language models for post-transcription correction in contact centers', 'koilakuntla24_interspeech', 'anchor downstream context brand correct compounded skilled transcription analytics pinpoint'], ['Dmitrii Obukhov, Marcel de Korte, Andrey Adaschik', 'ATTEST: an analytics tool for the testing and evaluation of speech technologies', 'obukhov24_interspeech', 'metric released powerful framework acknowledging need user-friendly alongside encourage large'], ['Margot Masson, Erfan A. Shams, Iona Gessinger, Julie Carson-Berndsen', 'PhoneViz: exploring alignments at a glance', 'masson24_interspeech', 'phone spanish-accented chart substituted showcase helping user ipa phonetic concrete'], ['Clément Pages, Hervé Bredin', 'Gryannote open-source speaker diarization labeling tool', 'pages24_interspeech', 'pyannote pipeline export ecosystem upload hyper-parameters customize visualize component custom'], ['Kai Liu, Ziqing Du, Zhou Huan, Xucheng Wan, Naijun Zheng', 'Real-time scheme for rapid extraction of speaker embeddings in challenging recording conditions', 'liu24s_interspeech', 'speaker-related enrollment target task embedding pristine three-stage realm compromising non-target'], ['Meenakshi Sirigiraju, Arjun Rajasekar, Abhishikth Meejuri, Chiranjeevi Yarra', 'IIITH Ucchar e-Sudharak: an automatic English pronunciation corrector for school-going children with a teacher in the loop', 'sirigiraju24_interspeech', 'tool skill practice student word-level gap sentence language catering empowers'], ['Boon Peng Yap, Kok Liang Tan, Zhenghao Li, Rong Tong', 'Speech enabled visual acuity test', 'yap24_interspeech', 'user posture eye system self-paced sight communicates automatically assess private'], ['Mayuko Aiba, Daisuke Saito, Nobuaki Minematsu', 'A ChatGPT-based oral Q&A practice system for first-time student participants in international conferences', 'aiba24_interspeech', 'question chatgpt configuration affirmed generation reference upload uploaded orally practicing'], ['Szu-Yu Chen, Tien-Hong Lo, Yao-Ting Sung, Ching-Yu Tseng, Berlin Chen', 'TEEMI: a speaking practice tool for L2 English learners', 'chen24aa_interspeech', 'foreign user automated assessment globalization surpassed laptop tablet language english-speaking'], ['Karthik Venkat Sridaran, Raja Praveen, Reuben T Varghese, Ajish K Abraham, Shankar R, Winnie Rachel Cheri
11146an', 'Visual scene display application for augmentative and alternative communication', 'sridaran24_interspeech', 'aac individual development ability language enhance icon webpage ensured limited'], ['Ikuyo Masuda-Katsuse, Ayako Shirose', 'CALL system using pitch-accent feature representations reflecting listeners’ subjective adequacy', 'masudakatsuse24_interspeech', 'quantitatively accepted implement evaluating learner japanese accent native automatically pitch'], ['Jonathan L Preston, Nina R Benway, Nathan Prestopnik, Nathan Preston', 'The speech motor chaining web app for speech motor learning', 'preston24_interspeech', 'principle invoke apps design computerized underlie illustrates therapy highlighting game'], ['Charlotte Yoder, Karrie Karahalios, Mark Hasegawa-Johnson, Shreyansh Agrawal', 'Visualization for improving foreign language pronunciation', 'yoder24_interspeech', 'chart vowel coherently seldom calibrated tutorial execution learning page emphasized'], ['Nhan Phan, Anna von Zansen, Maria Kautonen, Tamás Grósz, Mikko Kurimo', 'CaptainA self-study mobile app for practising speaking: task completion assessment and feedback with generative AI', 'phan24b_interspeech', 'language learner finnish providing picture-based automatic visual nlg grading asa'], ['Mohd Mujtaba Akhtar,  Girish, Orchid Chetia Phukan, Muskaan Singh', 'NeuRO: an application for code-switched autism detection in children', 'akhtar24_interspeech', 'asd individual disorder conversation communication code-switch posing challenge repetitive code-switching'], ['Orchid Chetia Phukan, Sarthak Jain, Shubham Singh, Muskaan Singh, Arun Balaji Buduru, Rajesh Sharma', 'ComFeAT: combination of neural and spectral features for improved depression detection', 'phukan24c_interspeech', 'ptms extracted sota decline previous trained individually speech-based paralinguistic cnn'], ['Isabel Trancoso', 'Towards Responsible Speech Processing', 'trancoso24_interspeech', 'pillar sustainability fairness explainability urgent uniquely nonetheless attempting inform inclusion'], ['Shoko Araki', 'Frontier of Frontend for Conversational Speech Processing', 'araki24_interspeech', 'conversation technology distant diarization evolution decade talk progress microphone enhancement'], ['Elmar Noeth', 'Analysis of Pathological Speech – Pitfalls along the Way', 'noeth24_interspeech', 'congenital aspect biomarker explainability defect less various privacy disease motivation'], ['Barbara Tillmann', 'Perception of music and speech: Focus on rhythm processing', 'tillmann24_interspeech', 'rhythmic research language cognitive disorder early revealed dyslexia hypothesis neuroscience']],
11147                    stateSave: true,
11148                    columnDefs: [
11149                        { targets: [0],
11150                          className: 'dt-left',
11151                          "mRender": function (data, type, full) {
11152                              return '<a class="w3-text" href="' + full[2] + '.html' + '">' + full[1] + '<br><span class="w3-text w3-text-theme">' + full[0] + '</span></a>';
11153                          }
11154                        },
11155                        { targets: [1, 2, 3],
11156                          visible: false,
11157                        },
11158                    ],
11159                    "lengthMenu": [7, 10, 20, 50, 100, 200, 500],
11160                    "pageLength": 50,
11161                    "order": [[ 0, 'asc' ]],
11162                    scrollY:        '60vh',
11163                    "dom": '<"top"l>rft<"bottom"ip><"clear">',
11164                    "pagingType": "full_numbers",
11165                    paging:         true
11166                });
11167
11168            });
11169
11170        </script>
11170
11171
11172    </body>
11173</html>

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.