{"url":"/task/speech-enhancement","name":"Speech Enhancement","slug":"speech-enhancement","description_markdown":"**Speech Enhancement** is a signal processing task that involves improving the quality of speech signals captured under noisy or degraded conditions. The goal of speech enhancement is to make speech signals clearer, more intelligible, and more pleasant to listen to, which can be used for various applications such as voice recognition, teleconferencing, and hearing aids. A representative Github project with online demo : [ClearerVoice-Studio](https://github.com/modelscope/ClearerVoice-Studio).\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [A Fully Convolutional Neural Network For Speech Enhancement](https://arxiv.org/pdf/1609.07132v1.pdf) )</span>","categories":[{"name":"Audio","url":"/area/audio"},{"name":"Speech","url":"/area/speech"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":982,"papers_with_code":280,"benchmarks":17,"benchmark_tables_in_archive":18,"benchmark_tables_shown":18,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":24,"subtasks":4,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/speech-enhancement-on-demand","slug":"speech-enhancement-on-demand","dataset":"VoiceBank + DEMAND","dataset_url":"/dataset/demand","rows_in_archive":42,"metrics":["PESQ (wb)","CBAK","COVL","CSIG","STOI","ESTOI","SSNR","SI-SDR","Para. (M)"],"first_row_in_archive_order":{"model":"ROSE-CD(PESQ)","paper_title":"Robust One-step Speech Enhancement via Consistency Distillation","paper_url":"/paper/robust-one-step-speech-enhancement-via-1","paper_date":"2025-07-08","arxiv_id":"2507.05688","code_links":[{"title":"LiangXu123/Robust-One-step-Speech-Enhancement-via-Consistency-Distillation-ROSE-CD-","url":"https://github.com/LiangXu123/Robust-One-step-Speech-Enhancement-via-Consistency-Distillation-ROSE-CD-"}],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-deep-noise-suppression","slug":"speech-enhancement-on-deep-noise-suppression","dataset":"Deep Noise Suppression (DNS) Challenge","dataset_url":"/dataset/deep-noise-suppression-2020","rows_in_archive":36,"metrics":["PESQ-WB","SI-SDR-WB","STOI","PESQ-NB","SI-SDR-NB","Number of parameters (M)","FLOPS (G)","ESTOI","SSNR"],"first_row_in_archive_order":{"model":"ZipEnhancer (M)","paper_title":null,"paper_url":null,"paper_date":"","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-chime-3","slug":"speech-enhancement-on-chime-3","dataset":"CHiME-3","dataset_url":null,"rows_in_archive":6,"metrics":["SDR","PESQ","ΔPESQ","STOI"],"first_row_in_archive_order":{"model":"Inter-Channel Conv-TasNet","paper_title":"Inter-channel Conv-TasNet for multichannel speech enhancement","paper_url":"/paper/inter-channel-conv-tasnet-for-multichannel","paper_date":"2021-11-08","arxiv_id":"2111.04312","code_links":[],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-ears-wham","slug":"speech-enhancement-on-ears-wham","dataset":"EARS-WHAM","dataset_url":"/dataset/ears-wham","rows_in_archive":6,"metrics":["PESQ-WB","SI-SDR","ESTOI","SIGMOS","DNSMOS","POLQA"],"first_row_in_archive_order":{"model":"Schrödinger Bridge (PESQ loss)","paper_title":"Investigating Training Objectives for Generative Speech Enhancement","paper_url":"/paper/investigating-training-objectives-for","paper_date":"2024-09-16","arxiv_id":"2409.10753","code_links":[{"title":"sp-uhh/sgmse","url":"https://github.com/sp-uhh/sgmse"}],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-easycom","slug":"speech-enhancement-on-easycom","dataset":"EasyCom","dataset_url":"/dataset/easycom","rows_in_archive":6,"metrics":["PESQ","STOI","ViSQOL","HASQI","Audio Quality MOS","SDR","ESTOI","HASPI","SI-SDR","SIIB","SNR","SegSNR"],"first_row_in_archive_order":{"model":"MaxDI (Baseline)","paper_title":"EasyCom: An Augmented Reality Dataset to Support Algorithms for Easy Communication in Noisy Environments","paper_url":"/paper/easycom-an-augmented-reality-dataset-to","paper_date":"2021-07-09","arxiv_id":"2107.04174","code_links":[{"title":"facebookresearch/EasyComDataset","url":"https://github.com/facebookresearch/EasyComDataset"}],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-interspeech-2020-deep","slug":"speech-enhancement-on-interspeech-2020-deep","dataset":"DNS Challenge","dataset_url":"/dataset/deep-noise-suppression-2020","rows_in_archive":5,"metrics":["PESQ-NB","PESQ-WB"],"first_row_in_archive_order":{"model":"ZipEnhancer\n(M)","paper_title":null,"paper_url":null,"paper_date":"","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-vb-demandex","slug":"speech-enhancement-on-vb-demandex","dataset":"VB-DemandEx","dataset_url":"/dataset/vb-demandex","rows_in_archive":4,"metrics":["ESTOI","Number of parameters (M)","PESQ (wb)","SI-SDR","SSNR"],"first_row_in_archive_order":{"model":"MambAttention","paper_title":"MambAttention: Mamba with Multi-Head Attention for Generalizable Single-Channel Speech Enhancement","paper_url":"/paper/mambattention-mamba-with-multi-head-attention","paper_date":"2025-07-01","arxiv_id":"2507.00966","code_links":[{"title":"nikolaikyhne/xlstm-senet","url":"https://github.com/nikolaikyhne/xlstm-senet"},{"title":"NikolaiKyhne/MambAttention","url":"https://github.com/NikolaiKyhne/MambAttention"}],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-whamr","slug":"speech-enhancement-on-whamr","dataset":"WHAMR!","dataset_url":"/dataset/whamr","rows_in_archive":4,"metrics":["PESQ","SI-SDR","ΔPESQ","SI-SNR","SDR"],"first_row_in_archive_order":{"model":"SepFormer","paper_title":"Exploring Self-Attention Mechanisms for Speech Separation","paper_url":"/paper/on-using-transformers-for-speech-separation","paper_date":"2022-02-06","arxiv_id":"2202.02884","code_links":[{"title":"speechbrain/speechbrain","url":"https://github.com/speechbrain/speechbrain"}],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-wsj0-demand-rnnoise","slug":"speech-enhancement-on-wsj0-demand-rnnoise","dataset":"WSJ0 + DEMAND + RNNoise","dataset_url":null,"rows_in_archive":3,"metrics":["PESQ-NB"],"first_row_in_archive_order":{"model":"DCUNet-MC","paper_title":"Monaural Speech Enhancement with Complex Convolutional Block Attention Module and Joint Time Frequency Losses","paper_url":"/paper/monaural-speech-enhancement-with-complex","paper_date":"2021-02-03","arxiv_id":"2102.01993","code_links":[{"title":"modelscope/ClearerVoice-Studio","url":"https://github.com/modelscope/ClearerVoice-Studio"},{"title":"alibabasglab/frcrn","url":"https://github.com/alibabasglab/frcrn"}],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-realman","slug":"speech-enhancement-on-realman","dataset":"RealMAN","dataset_url":"/dataset/realman","rows_in_archive":2,"metrics":["DNSMOS","DNSMOS BAK","DNSMOS OVRL","DNSMOS SIG","PESQ-WB"],"first_row_in_archive_order":{"model":"CleanMel-L-map","paper_title":"CleanMel: Mel-Spectrogram Enhancement for Improving Both Speech Quality and ASR","paper_url":"/paper/cleanmel-mel-spectrogram-enhancement-for","paper_date":"2025-02-27","arxiv_id":"2502.20040","code_links":[{"title":"Audio-WestlakeU/CleanMel","url":"https://github.com/Audio-WestlakeU/CleanMel"}],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-voicebank-demand-2","slug":"speech-enhancement-on-voicebank-demand-2","dataset":"VoiceBank+DEMAND","dataset_url":"/dataset/voice-bank-demand","rows_in_archive":2,"metrics":["PESQ","DNSMOS","DNSMOS BAK","DNSMOS OVRL","DNSMOS SIG","ESTOI","SI-SDR","PESQ (wb)"],"first_row_in_archive_order":{"model":"ROSE-CD","paper_title":"Robust One-step Speech Enhancement via Consistency Distillation","paper_url":"/paper/robust-one-step-speech-enhancement-via-1","paper_date":"2025-07-08","arxiv_id":"2507.05688","code_links":[{"title":"LiangXu123/Robust-One-step-Speech-Enhancement-via-Consistency-Distillation-ROSE-CD-","url":"https://github.com/LiangXu123/Robust-One-step-Speech-Enhancement-via-Consistency-Distillation-ROSE-CD-"}],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-demand-1","slug":"speech-enhancement-on-demand-1","dataset":"DEMAND","dataset_url":null,"rows_in_archive":1,"metrics":["CBAK","COVL","CSIG","PESQ"],"first_row_in_archive_order":{"model":"Wave-U-Net","paper_title":"Improved Speech Enhancement with the Wave-U-Net","paper_url":"/paper/improved-speech-enhancement-with-the-wave-u","paper_date":"2018-11-27","arxiv_id":"1811.11307","code_links":[{"title":"craigmacartney/Wave-U-Net-For-Speech-Enhancement","url":"https://github.com/craigmacartney/Wave-U-Net-For-Speech-Enhancement"},{"title":"pheepa/DCUnet","url":"https://github.com/pheepa/DCUnet"},{"title":"MattSegal/speech-enhancement","url":"https://github.com/MattSegal/speech-enhancement"}],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-grid-corpus-mixed","slug":"speech-enhancement-on-grid-corpus-mixed","dataset":"GRID corpus (mixed-speech)","dataset_url":"/dataset/grid","rows_in_archive":1,"metrics":["PESQ"],"first_row_in_archive_order":{"model":"Audio-Visual concat-ref","paper_title":"Face Landmark-based Speaker-Independent Audio-Visual Speech Enhancement in Multi-Talker Environments","paper_url":"/paper/face-landmark-based-speaker-independent-audio","paper_date":"2018-11-06","arxiv_id":"1811.02480","code_links":[{"title":"dr-pato/audio_visual_speech_enhancement","url":"https://github.com/dr-pato/audio_visual_speech_enhancement"}],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-librispeech","slug":"speech-enhancement-on-librispeech","dataset":"LibriSpeechDuplicate","dataset_url":null,"rows_in_archive":1,"metrics":["Audio Quality MOS"],"first_row_in_archive_order":{"model":"SE-MelGAN","paper_title":"SE-MelGAN -- Speaker Agnostic Rapid Speech Enhancement","paper_url":"/paper/se-melgan-speaker-agnostic-rapid-speech","paper_date":"2020-06-13","arxiv_id":"2006.07637","code_links":[],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-spatialized-dns","slug":"speech-enhancement-on-spatialized-dns","dataset":"spatialized DNS challenge","dataset_url":null,"rows_in_archive":1,"metrics":["PESQ","SI-SDR","STOI"],"first_row_in_archive_order":{"model":"DeFT-AN","paper_title":"DeFT-AN: Dense Frequency-Time Attentive Network for Multichannel Speech Enhancement","paper_url":"/paper/deft-an-dense-frequency-time-attentive","paper_date":"2022-12-15","arxiv_id":"2212.07570","code_links":[{"title":"donghoney0416/DeFT-AN","url":"https://github.com/donghoney0416/DeFT-AN"}],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-tcd-timit-corpus-mixed","slug":"speech-enhancement-on-tcd-timit-corpus-mixed","dataset":"TCD-TIMIT corpus (mixed-speech)","dataset_url":"/dataset/timit","rows_in_archive":1,"metrics":["PESQ"],"first_row_in_archive_order":{"model":"Audio-Visual concat-ref","paper_title":"Face Landmark-based Speaker-Independent Audio-Visual Speech Enhancement in Multi-Talker Environments","paper_url":"/paper/face-landmark-based-speaker-independent-audio","paper_date":"2018-11-06","arxiv_id":"1811.02480","code_links":[{"title":"dr-pato/audio_visual_speech_enhancement","url":"https://github.com/dr-pato/audio_visual_speech_enhancement"}],"syntology":null}},{"leaderboard":"/sota/speech-enhancement-on-wham","slug":"speech-enhancement-on-wham","dataset":"WHAM!","dataset_url":"/dataset/wham","rows_in_archive":1,"metrics":["PESQ","SDR","SI-SNR"],"first_row_in_archive_order":{"model":"SepFormer","paper_title":"Exploring Self-Attention Mechanisms for Speech Separation","paper_url":"/paper/on-using-transformers-for-speech-separation","paper_date":"2022-02-06","arxiv_id":"2202.02884","code_links":[{"title":"speechbrain/speechbrain","url":"https://github.com/speechbrain/speechbrain"}],"syntology":null}},{"leaderboard":null,"slug":"speech-enhancement-on-dns-4","dataset":"DNS-4","dataset_url":null,"rows_in_archive":0,"metrics":["DNSMOS BAK","DNSMOS OVRL","DNSMOS SIG"],"first_row_in_archive_order":null}],"datasets":[{"url":"/dataset/librimix","name":"LibriMix","full_name":"","num_papers_in_archive":122},{"url":"/dataset/wham","name":"WHAM!","full_name":"WSJ0 Hipster Ambient Mixtures","num_papers_in_archive":114},{"url":"/dataset/ava","name":"AVA","full_name":"Atomic Visual Actions","num_papers_in_archive":113},{"url":"/dataset/whamr","name":"WHAMR!","full_name":"WHAM! with synthetic reverberated sources","num_papers_in_archive":57},{"url":"/dataset/reverb-challenge","name":"ReVerb Challenge","full_name":"REverberant Voice Enhancement and Recognition Benchmark","num_papers_in_archive":55},{"url":"/dataset/demand","name":"VoiceBank + DEMAND","full_name":"Noisy speech database for training speech enhancement algorithms and TTS models","num_papers_in_archive":53},{"url":"/dataset/deep-noise-suppression-2020","name":"DNS Challenge","full_name":"Deep Noise Suppression Challenge","num_papers_in_archive":48},{"url":"/dataset/chime-5","name":"CHiME-5","full_name":"CHiME Speech Separation and Recognition Challenge","num_papers_in_archive":42},{"url":"/dataset/timit","name":"TIMIT","full_name":"TIMIT Acoustic-Phonetic Continuous Speech Corpus","num_papers_in_archive":31},{"url":"/dataset/ava-activespeaker","name":"AVA-ActiveSpeaker","full_name":"","num_papers_in_archive":22},{"url":"/dataset/easycom","name":"EasyCom","full_name":"","num_papers_in_archive":22},{"url":"/dataset/voice-bank-demand","name":"VoiceBank+DEMAND","full_name":"","num_papers_in_archive":16},{"url":"/dataset/l3das22","name":"L3DAS22","full_name":"","num_papers_in_archive":13},{"url":"/dataset/ears-wham","name":"EARS-WHAM","full_name":"","num_papers_in_archive":11},{"url":"/dataset/grid","name":"GRID Dataset","full_name":"","num_papers_in_archive":10},{"url":"/dataset/l3das21","name":"L3DAS21","full_name":"","num_papers_in_archive":6},{"url":"/dataset/realman","name":"RealMAN","full_name":"A Real-Recorded and Annotated Microphone Array Dataset for Dynamic Speech Enhancement and Localization","num_papers_in_archive":5},{"url":"/dataset/cas-vsr-s101","name":"CAS-VSR-S101","full_name":"","num_papers_in_archive":1},{"url":"/dataset/gneutralspeech-female","name":"GneutralSpeech Female","full_name":"","num_papers_in_archive":1},{"url":"/dataset/gneutralspeech-male","name":"GneutralSpeech Male","full_name":"","num_papers_in_archive":1},{"url":"/dataset/mdrt","name":"mDRT","full_name":"Multilingual Diagnostic Rhyme Test","num_papers_in_archive":1},{"url":"/dataset/nisqa-speech-quality-corpus","name":"NISQA Speech Quality Corpus","full_name":"","num_papers_in_archive":1},{"url":"/dataset/vb-demandex","name":"VB-DemandEx","full_name":"","num_papers_in_archive":1},{"url":"/dataset/whamr-ext","name":"WHAMR_ext","full_name":"WHAMR_ext","num_papers_in_archive":1}],"subtasks":[{"url":"/task/bandwidth-extension","name":"Bandwidth Extension"},{"url":"/task/packet-loss-concealment","name":"Packet Loss Concealment"},{"url":"/task/speech-dereverberation","name":"Speech Dereverberation"},{"url":"/task/speech-intelligibility-evaluation","name":"Speech Intelligibility Evaluation"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":280,"tagged_in_all":982,"items":[{"url":"/paper/proximal-policy-optimization-algorithms","title":"Proximal Policy Optimization Algorithms","date":"2017-07-20","arxiv_id":"1707.06347","repositories_listed":188,"syntology":{"n":176,"n_ran":99,"n_unverified":77,"n_pointer_only":94}},{"url":"/paper/perceptual-losses-for-real-time-style","title":"Perceptual Losses for Real-Time Style Transfer and Super-Resolution","date":"2016-03-27","arxiv_id":"1603.08155","repositories_listed":80,"syntology":{"n":46,"n_ran":13,"n_unverified":33,"n_pointer_only":5}},{"url":"/paper/segan-speech-enhancement-generative","title":"SEGAN: Speech Enhancement Generative Adversarial Network","date":"2017-03-28","arxiv_id":"1703.09452","repositories_listed":21,"syntology":{"n":39,"n_ran":3,"n_unverified":36,"n_pointer_only":4}},{"url":"/paper/tasnet-surpassing-ideal-time-frequency","title":"Conv-TasNet: Surpassing Ideal Time-Frequency Magnitude Masking for Speech Separation","date":"2018-09-20","arxiv_id":"1809.07454","repositories_listed":17,"syntology":{"n":36,"n_ran":7,"n_unverified":29,"n_pointer_only":16}},{"url":"/paper/zero-reference-deep-curve-estimation-for-low","title":"Zero-Reference Deep Curve Estimation for Low-Light Image Enhancement","date":"2020-01-19","arxiv_id":"2001.06826","repositories_listed":13,"syntology":{"n":16,"n_ran":1,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/phase-aware-speech-enhancement-with-deep-1","title":"Phase-aware Speech Enhancement with Deep Complex U-Net","date":"2019-03-07","arxiv_id":"1903.03107","repositories_listed":9,"syntology":{"n":10,"n_ran":3,"n_unverified":7,"n_pointer_only":2}},{"url":"/paper/speecht5-unified-modal-encoder-decoder-pre","title":"SpeechT5: Unified-Modal Encoder-Decoder Pre-Training for Spoken Language Processing","date":"2021-10-14","arxiv_id":"2110.07205","repositories_listed":6,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/soundstream-an-end-to-end-neural-audio-codec","title":"SoundStream: An End-to-End Neural Audio Codec","date":"2021-07-07","arxiv_id":"2107.03312","repositories_listed":6,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/fullsubnet-a-full-band-and-sub-band-fusion","title":"FullSubNet: A Full-Band and Sub-Band Fusion Model for Real-Time Single-Channel Speech Enhancement","date":"2020-10-29","arxiv_id":"2010.15508","repositories_listed":6,"syntology":{"n":21,"n_ran":1,"n_unverified":20,"n_pointer_only":0}},{"url":"/paper/a-fully-convolutional-neural-network-for","title":"A Fully Convolutional Neural Network for Speech Enhancement","date":"2016-09-22","arxiv_id":"1609.07132","repositories_listed":6,"syntology":null},{"url":"/paper/wavecrn-an-efficient-convolutional-recurrent","title":"WaveCRN: An Efficient Convolutional Recurrent Neural Network for End-to-end Speech Enhancement","date":"2020-04-06","arxiv_id":"2004.04098","repositories_listed":5,"syntology":null},{"url":"/paper/metricgan-generative-adversarial-networks","title":"MetricGAN: Generative Adversarial Networks based Black-box Metric Scores Optimization for Speech Enhancement","date":"2019-05-13","arxiv_id":"1905.04874","repositories_listed":5,"syntology":null},{"url":"/paper/voicefilter-targeted-voice-separation-by","title":"VoiceFilter: Targeted Voice Separation by Speaker-Conditioned Spectrogram Masking","date":"2018-10-11","arxiv_id":"1810.04826","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/rnnoise-ex-hybrid-speech-enhancement-system","title":"RNNoise-Ex: Hybrid Speech Enhancement System based on RNN and Spectral Features","date":"2021-05-25","arxiv_id":"2105.11813","repositories_listed":4,"syntology":null},{"url":"/paper/whispered-to-voiced-alaryngeal-speech","title":"Whispered-to-voiced Alaryngeal Speech Conversion with Generative Adversarial Networks","date":"2018-08-31","arxiv_id":"1808.10687","repositories_listed":4,"syntology":null},{"url":"/paper/hifi-a-unified-framework-for-neural-vocoding","title":"HiFi++: a Unified Framework for Bandwidth Extension and Speech Enhancement","date":"2022-03-24","arxiv_id":"2203.13086","repositories_listed":3,"syntology":{"n":12,"n_ran":9,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/metricgan-an-improved-version-of-metricgan","title":"MetricGAN+: An Improved Version of MetricGAN for Speech Enhancement","date":"2021-04-08","arxiv_id":"2104.03538","repositories_listed":3,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/real-time-speech-enhancement-in-the-waveform","title":"Real Time Speech Enhancement in the Waveform Domain","date":"2020-06-23","arxiv_id":"2006.12847","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/weighted-speech-distortion-losses-for-neural-1","title":"Weighted Speech Distortion Losses for Neural-network-based Real-time Speech Enhancement","date":"2020-02-12","arxiv_id":"2001.10601","repositories_listed":3,"syntology":null},{"url":"/paper/spleeter-a-fast-and-state-of-the-art-music","title":"Spleeter: A Fast And State-of-the Art Music Source Separation Tool With Pre-trained Models","date":"2019-11-04","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/rvad-an-unsupervised-segment-based-robust","title":"rVAD: An Unsupervised Segment-Based Robust Voice Activity Detection Method","date":"2019-06-09","arxiv_id":"1906.03588","repositories_listed":3,"syntology":null},{"url":"/paper/fast-multichannel-source-separation-based-on","title":"Fast Multichannel Source Separation Based on Jointly Diagonalizable Spatial Covariance Matrices","date":"2019-03-08","arxiv_id":"1903.03237","repositories_listed":3,"syntology":null},{"url":"/paper/improved-speech-enhancement-with-the-wave-u","title":"Improved Speech Enhancement with the Wave-U-Net","date":"2018-11-27","arxiv_id":"1811.11307","repositories_listed":3,"syntology":null},{"url":"/paper/language-and-noise-transfer-in-speech","title":"Language and Noise Transfer in Speech Enhancement Generative Adversarial Network","date":"2017-12-18","arxiv_id":"1712.06340","repositories_listed":3,"syntology":null},{"url":"/paper/mambattention-mamba-with-multi-head-attention","title":"MambAttention: Mamba with Multi-Head Attention for Generalizable Single-Channel Speech Enhancement","date":"2025-07-01","arxiv_id":"2507.00966","repositories_listed":2,"syntology":null},{"url":"/paper/sonicsim-a-customizable-simulation-platform","title":"SonicSim: A customizable simulation platform for speech processing in moving sound source scenarios","date":"2024-10-02","arxiv_id":"2410.01481","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_unverified":3,"n_pointer_only":11}},{"url":"/paper/hold-me-tight-stable-encoder-decoder-design","title":"Hold Me Tight: Stable Encoder-Decoder Design for Speech Enhancement","date":"2024-08-30","arxiv_id":"2408.17358","repositories_listed":2,"syntology":null},{"url":"/paper/ears-an-anechoic-fullband-speech-dataset","title":"EARS: An Anechoic Fullband Speech Dataset Benchmarked for Speech Enhancement and Dereverberation","date":"2024-06-10","arxiv_id":"2406.06185","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/lauragpt-listen-attend-understand-and","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","date":"2023-10-07","arxiv_id":"2310.04673","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/multi-dimensional-speech-quality-assessment","title":"Multi-dimensional Speech Quality Assessment in Crowdsourcing","date":"2023-09-14","arxiv_id":"2309.07385","repositories_listed":2,"syntology":null}],"syntology_records":16,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}