@inproceedings{f980c92b8f704978a237fd873df97ae8,
title = "Multi-scale TCN: Exploring better temporal DNN model for causal speech enhancement",
abstract = "Capturing the temporal dependence of speech signals is of great importance for numerous speech related tasks. This paper proposes a more effective temporal modeling method for causal speech enhancement system. We design a forward stacked temporal convolutional network (TCN) model which exploits multi-scale temporal analysis in each residual block. This model incorporates a multi-scale dilated convolution to better track the target speech through its context information from past frames. Applying multi-target learning of log power spectrum (LPS) and ideal ratio mask (IRM) further improves model robustness, due to the complementarity among the tasks. Experimental results show that the proposed TCN model not only performs better speech reconstruction ability in terms of speech quality and speech intelligibility, but also has smaller model size than that of long short-term memory (LSTM) network and the gated recurrent units (GRU) network.",
keywords = "Multi-objective learning, Multi-scale, Speech enhancement, Temporal convolutional network",
author = "Lu Zhang and Mingjiang Wang",
note = "Publisher Copyright: Copyright {\textcopyright} 2020 ISCA; 21st Annual Conference of the International Speech Communication Association, INTERSPEECH 2020 ; Conference date: 25-10-2020 Through 29-10-2020",
year = "2020",
doi = "10.21437/Interspeech.2020-1104",
language = "英语",
isbn = "9781713820697",
series = "Proceedings of the Annual Conference of the International Speech Communication Association, INTERSPEECH",
publisher = "International Speech Communication Association",
pages = "2672--2676",
booktitle = "Interspeech 2020",
address = "法国",
}