{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-enhancement/papers/5","list_of":"/task/speech-enhancement","task":"Speech Enhancement","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":10,"rows_per_page":100,"rows":[401,500],"of":982,"counts":{"archive_papers_tagged":982,"with_a_code_link":280,"where_syntology_ran_a_sample":54,"not_listed_spam_title":0,"listed":982,"listed_where_code_ran":54,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":45,"every_run_a_failure_of_syntologys_instrument":9,"listed_with_a_run_with_no_instrument_failure":45,"listed_every_run_a_failure_of_syntologys_instrument":9,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-enhancement","prev":"/task/speech-enhancement/papers/4","next":"/task/speech-enhancement/papers/6","papers":[{"url":null,"slug":"the-effect-of-training-dataset-size-on","title":"The Effect of Training Dataset Size on Discriminative and Diffusion-Based Speech Enhancement Systems","date":"2024-06-10","arxiv_id":"2406.06160","repositories_listed":0,"syntology":null},{"url":null,"slug":"thunder-unified-regression-diffusion-speech","title":"Thunder : Unified Regression-Diffusion Speech Enhancement with a Single Reverse Step using Brownian Bridge","date":"2024-06-10","arxiv_id":"2406.06139","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-of-noise-robustness-for-flow","title":"An Investigation of Noise Robustness for Flow-Matching-Based Zero-Shot TTS","date":"2024-06-09","arxiv_id":"2406.05699","repositories_listed":0,"syntology":null},{"url":null,"slug":"urgent-challenge-universality-robustness-and","title":"URGENT Challenge: Universality, Robustness, and Generalizability For Speech Enhancement","date":"2024-06-07","arxiv_id":"2406.04660","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexible-multichannel-speech-enhancement-for","title":"Flexible Multichannel Speech Enhancement for Noise-Robust Frontend","date":"2024-06-06","arxiv_id":"2406.04552","repositories_listed":0,"syntology":null},{"url":null,"slug":"helsinki-speech-challenge-2024","title":"Helsinki Speech Challenge 2024","date":"2024-06-06","arxiv_id":"2406.04123","repositories_listed":0,"syntology":null},{"url":null,"slug":"pldnet-pld-guided-lightweight-deep-network","title":"PLDNet: PLD-Guided Lightweight Deep Network Boosted by Efficient Attention for Handheld Dual-Microphone Speech Enhancement","date":"2024-06-06","arxiv_id":"2406.03899","repositories_listed":0,"syntology":null},{"url":null,"slug":"reference-channel-selection-by-multi-channel","title":"Reference Channel Selection by Multi-Channel Masking for End-to-End Multi-Channel Speech Enhancement","date":"2024-06-05","arxiv_id":"2406.03228","repositories_listed":0,"syntology":null},{"url":"/paper/the-pesqetarian-on-the-relevance-of-goodhart","slug":"the-pesqetarian-on-the-relevance-of-goodhart","title":"The PESQetarian: On the Relevance of Goodhart's Law for Speech Enhancement","date":"2024-06-05","arxiv_id":"2406.03460","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-enhancement-deep-learning-architecture","title":"Speech enhancement deep-learning architecture for efficient edge processing","date":"2024-05-27","arxiv_id":"2405.16834","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-real-time-accent","title":"Non-autoregressive real-time Accent Conversion model with voice cloning","date":"2024-05-21","arxiv_id":"2405.13162","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-luganda-text-to-speech-model-from","title":"Building a Luganda Text-to-Speech Model From Crowdsourced Data","date":"2024-05-16","arxiv_id":"2405.10211","repositories_listed":0,"syntology":null},{"url":null,"slug":"monaural-speech-enhancement-on-drone-via","title":"Monaural speech enhancement on drone via Adapter based transfer learning","date":"2024-05-16","arxiv_id":"2405.10022","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-speech-enhancement-systems-through","title":"Evaluating Speech Enhancement Systems Through Listening Effort","date":"2024-05-13","arxiv_id":"2405.07641","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-multichannel-deep-speech","title":"Real-time multichannel deep speech enhancement in hearing aids: Comparing monaural and binaural processing in complex acoustic scenarios","date":"2024-05-03","arxiv_id":"2405.01967","repositories_listed":0,"syntology":null},{"url":null,"slug":"tramba-a-hybrid-transformer-and-mamba","title":"TRAMBA: A Hybrid Transformer and Mamba Architecture for Practical Audio and Bone Conduction Speech Super Resolution and Enhancement on Mobile and Wearable Platforms","date":"2024-05-02","arxiv_id":"2405.01242","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-low-latency-joint-speech-transmission","title":"Deep low-latency joint speech transmission and enhancement over a gaussian channel","date":"2024-04-30","arxiv_id":"2404.19375","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-processing-distortions","title":"Rethinking Processing Distortions: Disentangling the Impact of Speech Enhancement Errors on Speech Recognition Performance","date":"2024-04-23","arxiv_id":"2404.14860","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-potential-of-data-driven","title":"Exploring the Potential of Data-Driven Spatial Audio Enhancement Using a Single-Channel Model","date":"2024-04-22","arxiv_id":"2404.14564","repositories_listed":0,"syntology":null},{"url":null,"slug":"trnet-two-level-refinement-network-leveraging","title":"TRNet: Two-level Refinement Network leveraging Speech Enhancement for Noise Robust Speech Emotion Recognition","date":"2024-04-19","arxiv_id":"2404.12979","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-high-performance-bark-scale-neural","title":"Efficient High-Performance Bark-Scale Neural Network for Residual Echo and Noise Suppression","date":"2024-04-08","arxiv_id":"2404.11621","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-artificial-intelligence-algorithms","title":"Artificial Intelligence for Cochlear Implants: Review of Strategies, Challenges, and Perspectives","date":"2024-03-17","arxiv_id":"2403.15442","repositories_listed":0,"syntology":null},{"url":null,"slug":"superme-supervised-and-mixture-to-mixture-co","title":"SuperM2M: Supervised and Mixture-to-Mixture Co-Learning for Speech Enhancement and Noise-Robust ASR","date":"2024-03-15","arxiv_id":"2403.10271","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-closer-look-at-wav2vec2-embeddings-for-on","title":"A Closer Look at Wav2Vec2 Embeddings for On-Device Single-Channel Speech Enhancement","date":"2024-03-03","arxiv_id":"2403.01369","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-of-adapter-for-noise-robust","title":"Exploration of Adapter for Noise Robust Automatic Speech Recognition","date":"2024-02-28","arxiv_id":"2402.18275","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-enhancement-in-noisy","title":"Audio-Visual Speech Enhancement in Noisy Environments via Emotion-Based Contextual Cues","date":"2024-02-26","arxiv_id":"2402.16394","repositories_listed":0,"syntology":null},{"url":null,"slug":"sicrn-advancing-speech-enhancement-through","title":"SICRN: Advancing Speech Enhancement through State Space Model and Inplace Convolution Techniques","date":"2024-02-22","arxiv_id":"2402.14225","repositories_listed":0,"syntology":null},{"url":null,"slug":"mel-fullsubnet-mel-spectrogram-enhancement","title":"Mel-FullSubNet: Mel-Spectrogram Enhancement for Improving Both Speech Quality and ASR","date":"2024-02-21","arxiv_id":"2402.13511","repositories_listed":0,"syntology":null},{"url":null,"slug":"plugin-speech-enhancement-a-universal-speech","title":"Plugin Speech Enhancement: A Universal Speech Enhancement Framework Inspired by Dynamic Neural Network","date":"2024-02-20","arxiv_id":"2402.12746","repositories_listed":0,"syntology":null},{"url":null,"slug":"secp-a-speech-enhancement-based-curation","title":"SECP: A Speech Enhancement-Based Curation Pipeline For Scalable Acquisition Of Clean Speech","date":"2024-02-19","arxiv_id":"2402.12482","repositories_listed":0,"syntology":null},{"url":"/paper/speaking-in-wavelet-domain-a-simple-and","slug":"speaking-in-wavelet-domain-a-simple-and","title":"Speaking in Wavelet Domain: A Simple and Efficient Approach to Speed up Speech Diffusion Model","date":"2024-02-16","arxiv_id":"2402.10642","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/speaking-in-wavelet-domain-a-simple-and#ran","syntology_url":"https://syntology.ai/paper/2402.10642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10642"}},"official":null}},{"url":null,"slug":"diffusion-models-for-audio-restoration","title":"Diffusion Models for Audio Restoration","date":"2024-02-15","arxiv_id":"2402.09821","repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-of-the-l3das23-challenge-on-audio","title":"Overview of the L3DAS23 Challenge on Audio-Visual Extended Reality","date":"2024-02-14","arxiv_id":"2402.09245","repositories_listed":0,"syntology":null},{"url":null,"slug":"unrestricted-global-phase-bias-aware-single","title":"Unrestricted Global Phase Bias-Aware Single-channel Speech Enhancement with Conformer-based Metric GAN","date":"2024-02-13","arxiv_id":"2402.08252","repositories_listed":0,"syntology":null},{"url":null,"slug":"array-geometry-robust-attention-based-neural","title":"Array Geometry-Robust Attention-Based Neural Beamformer for Moving Speakers","date":"2024-02-05","arxiv_id":"2402.03058","repositories_listed":0,"syntology":null},{"url":"/paper/an-analysis-of-the-variance-of-diffusion","slug":"an-analysis-of-the-variance-of-diffusion","title":"An Analysis of the Variance of Diffusion-based Speech Enhancement","date":"2024-02-01","arxiv_id":"2402.00811","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-stereo-speech-enhancement-with","title":"Real-time Stereo Speech Enhancement with Spatial-Cue Preservation based on Dual-Path Structure","date":"2024-02-01","arxiv_id":"2402.00337","repositories_listed":0,"syntology":null},{"url":null,"slug":"speechcomposer-unifying-multiple-speech-tasks","title":"SpeechComposer: Unifying Multiple Speech Tasks with Prompt Composition","date":"2024-01-31","arxiv_id":"2401.18045","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-two-stage-framework-in-cross-spectrum","title":"A Two-Stage Framework in Cross-Spectrum Domain for Real-Time Speech Enhancement","date":"2024-01-19","arxiv_id":"2401.10494","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-study-on-the-impact-of-1","title":"An Empirical Study on the Impact of Positional Encoding in Transformer-based Monaural Speech Enhancement","date":"2024-01-18","arxiv_id":"2401.09686","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-speech-pre-emphasis-as-a-simple-and","title":"On Speech Pre-emphasis as a Simple and Inexpensive Method to Boost Speech Enhancement","date":"2024-01-17","arxiv_id":"2401.09315","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-robust-zero-shot-text-to-speech","title":"Noise-robust zero-shot text-to-speech synthesis conditioned on self-supervised speech-representation model with adapters","date":"2024-01-10","arxiv_id":"2401.05111","repositories_listed":0,"syntology":null},{"url":null,"slug":"fadi-aec-fast-score-based-diffusion-model","title":"FADI-AEC: Fast Score Based Diffusion Model Guided by Far-end Signal for Acoustic Echo Cancellation","date":"2024-01-08","arxiv_id":"2401.04283","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-multichannel-far-field-speech","title":"A unified multichannel far-field speech recognition system: combining neural beamforming with attention based end-to-end model","date":"2024-01-05","arxiv_id":"2401.02673","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-channel-speech-enhancement-using-2","title":"Single-channel speech enhancement using learnable loss mixup","date":"2023-12-20","arxiv_id":"2312.17255","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-real-time-multi-stage-speech-enhancement","title":"On real-time multi-stage speech enhancement systems","date":"2023-12-19","arxiv_id":"2312.12415","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-driven-multichannel-speech","title":"Attention-Driven Multichannel Speech Enhancement in Moving Sound Source Scenarios","date":"2023-12-17","arxiv_id":"2312.10756","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-representation-learning-based-speech","title":"A Deep Representation Learning-based Speech Enhancement Method Using Complex Convolution Recurrent Variational Autoencoder","date":"2023-12-15","arxiv_id":"2312.09620","repositories_listed":0,"syntology":null},{"url":null,"slug":"selm-speech-enhancement-using-discrete-tokens","title":"SELM: Speech Enhancement Using Discrete Tokens and Language Models","date":"2023-12-15","arxiv_id":"2312.09747","repositories_listed":0,"syntology":null},{"url":null,"slug":"ultra-low-complexity-deep-learning-based","title":"Ultra Low Complexity Deep Learning Based Noise Suppression","date":"2023-12-13","arxiv_id":"2312.08132","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-based-speech-enhancement-in-matched","title":"Diffusion-Based Speech Enhancement in Matched and Mismatched Conditions Using a Heun-Based Sampler","date":"2023-12-05","arxiv_id":"2312.02683","repositories_listed":0,"syntology":null},{"url":null,"slug":"head-orientation-estimation-with-distributed","title":"Head Orientation Estimation with Distributed Microphones Using Speech Radiation Patterns","date":"2023-12-04","arxiv_id":"2312.01808","repositories_listed":0,"syntology":null},{"url":null,"slug":"sefgan-harvesting-the-power-of-normalizing","title":"SEFGAN: Harvesting the Power of Normalizing Flows and GANs for Efficient High-Quality Speech Enhancement","date":"2023-12-04","arxiv_id":"2312.01744","repositories_listed":0,"syntology":null},{"url":null,"slug":"subspace-hybrid-mvdr-beamforming-for","title":"Subspace Hybrid MVDR Beamforming for Augmented Hearing","date":"2023-11-30","arxiv_id":"2311.18689","repositories_listed":0,"syntology":null},{"url":null,"slug":"lc4sv-a-denoising-framework-learning-to","title":"LC4SV: A Denoising Framework Learning to Compensate for Unseen Speaker Verification Models","date":"2023-11-28","arxiv_id":"2311.16604","repositories_listed":0,"syntology":null},{"url":null,"slug":"cheapnet-improving-light-weight-speech","title":"CheapNET: Improving Light-weight speech enhancement network by projected loss function","date":"2023-11-27","arxiv_id":"2311.15959","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-dual-attention-for-audio-visual","title":"Cooperative Dual Attention for Audio-Visual Speech Enhancement with Facial Cues","date":"2023-11-24","arxiv_id":"2311.14275","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparsity-driven-eeg-channel-selection-for","title":"Sparsity-Driven EEG Channel Selection for Brain-Assisted Speech Enhancement","date":"2023-11-22","arxiv_id":"2311.13436","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-does-end-to-end-speech-recognition","title":"How does end-to-end speech recognition training impact speech enhancement artifacts?","date":"2023-11-20","arxiv_id":"2311.11599","repositories_listed":0,"syntology":null},{"url":null,"slug":"se-territory-monaural-speech-enhancement","title":"SE Territory: Monaural Speech Enhancement Meets the Fixed Virtual Perceptual Space Mapping","date":"2023-11-03","arxiv_id":"2311.01679","repositories_listed":0,"syntology":null},{"url":null,"slug":"dpatd-dual-phase-audio-transformer-for","title":"DPATD: Dual-Phase Audio Transformer for Denoising","date":"2023-10-30","arxiv_id":"2310.19588","repositories_listed":0,"syntology":null},{"url":null,"slug":"scenario-aware-audio-visual-tf-gridnet-for","title":"Scenario-Aware Audio-Visual TF-GridNet for Target Speech Extraction","date":"2023-10-30","arxiv_id":"2310.19644","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-channel-speech-enhancement-by-colored","title":"Single channel speech enhancement by colored spectrograms","date":"2023-10-26","arxiv_id":"2310.17142","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-pre-training-for-speech-with-flow","title":"Generative Pre-training for Speech with Flow Matching","date":"2023-10-25","arxiv_id":"2310.16338","repositories_listed":0,"syntology":null},{"url":null,"slug":"lc-ttfs-towards-lossless-network-conversion","title":"LC-TTFS: Towards Lossless Network Conversion for Spiking Neural Networks with TTFS Coding","date":"2023-10-23","arxiv_id":"2310.14978","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-beamforming-for-speech-enhancement-and","title":"Deep Beamforming for Speech Enhancement and Speaker Localization with an Array Response-Aware Loss Function","date":"2023-10-19","arxiv_id":"2310.12837","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-speech-enhancement-and-separation","title":"Real-time Speech Enhancement and Separation with a Unified Deep Neural Network for Single/Dual Talker Scenarios","date":"2023-10-16","arxiv_id":"2310.10026","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-single-speech-enhancement-model-unifying","title":"A Single Speech Enhancement Model Unifying Dereverberation, Denoising, Speaker Counting, Separation, and Extraction","date":"2023-10-12","arxiv_id":"2310.08277","repositories_listed":0,"syntology":null},{"url":null,"slug":"magnitude-and-phase-aware-speech-enhancement","title":"Magnitude-and-phase-aware Speech Enhancement with Parallel Sequence Modeling","date":"2023-10-11","arxiv_id":"2310.07316","repositories_listed":0,"syntology":null},{"url":null,"slug":"psychoacoustic-challenges-of-speech","title":"Psychoacoustic Challenges Of Speech Enhancement On VoIP Platforms","date":"2023-10-11","arxiv_id":"2310.07161","repositories_listed":0,"syntology":null},{"url":null,"slug":"vsanet-real-time-speech-enhancement-based-on","title":"VSANet: Real-time Speech Enhancement Based on Voice Activity Detection and Causal Spatial Attention","date":"2023-10-11","arxiv_id":"2310.07295","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experiment-on-an-automated-literature","title":"An experiment on an automated literature survey of data-driven speech enhancement methods","date":"2023-10-10","arxiv_id":"2310.06260","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-exploration-of-task-decoupling-on-two","title":"An Exploration of Task-decoupling on Two-stage Neural Post Filter for Real-time Personalized Acoustic Echo Cancellation","date":"2023-10-07","arxiv_id":"2310.04715","repositories_listed":0,"syntology":null},{"url":null,"slug":"mbtfnet-multi-band-temporal-frequency-neural","title":"MBTFNet: Multi-Band Temporal-Frequency Neural Network For Singing Voice Enhancement","date":"2023-10-06","arxiv_id":"2310.04369","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-fused-deep-denoising-sound-coding-strategy","title":"A Fused Deep Denoising Sound Coding Strategy for Bilateral Cochlear Implants","date":"2023-10-02","arxiv_id":"2310.01122","repositories_listed":0,"syntology":null},{"url":null,"slug":"usee-unified-speech-enhancement-and-editing","title":"uSee: Unified Speech Enhancement and Editing with Conditional Diffusion Models","date":"2023-10-02","arxiv_id":"2310.00900","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-universal-speech-enhancement-for","title":"Toward Universal Speech Enhancement for Diverse Input Conditions","date":"2023-09-29","arxiv_id":"2309.17384","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-single-channel-speech-enhancement","title":"Does Single-channel Speech Enhancement Improve Keyword Spotting Accuracy? A Case Study","date":"2023-09-27","arxiv_id":"2309.16060","repositories_listed":0,"syntology":null},{"url":null,"slug":"multichannel-voice-trigger-detection-based-on","title":"Multichannel Voice Trigger Detection Based on Transform-average-concatenate","date":"2023-09-27","arxiv_id":"2309.16036","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoprep-an-automatic-preprocessing-framework","title":"AutoPrep: An Automatic Preprocessing Framework for In-the-Wild Speech Data","date":"2023-09-25","arxiv_id":"2309.13905","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-conditional-expectation-model-for","title":"DDTSE: Discriminative Diffusion Model for Target Speech Extraction","date":"2023-09-25","arxiv_id":"2309.13874","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-enhancement-with-frequency-domain-auto","title":"Speech enhancement with frequency domain auto-regressive modeling","date":"2023-09-24","arxiv_id":"2309.13537","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multiscale-autoencoder-msae-framework-for","title":"A Multiscale Autoencoder (MSAE) Framework for End-to-End Neural Network Speech Enhancement","date":"2023-09-21","arxiv_id":"2309.12121","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-complex-u-net-with-conformer-for-audio","title":"Deep Complex U-Net with Conformer for Audio-Visual Speech Enhancement","date":"2023-09-20","arxiv_id":"2309.11059","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-minimum-processing-beamforming-and-near","title":"Joint Minimum Processing Beamforming and Near-end Listening Enhancement","date":"2023-09-20","arxiv_id":"2309.11243","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-based-speech-enhancement-with-a","title":"Diffusion-based speech enhancement with a weighted generative-supervised learning loss","date":"2023-09-19","arxiv_id":"2309.10457","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-speech-enhancement-for-low-resource","title":"Exploring Speech Enhancement for Low-resource Speech Synthesis","date":"2023-09-19","arxiv_id":"2309.10795","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-ultrasound-tongue-images-for-1","title":"Incorporating Ultrasound Tongue Images for Audio-Visual Speech Enhancement","date":"2023-09-19","arxiv_id":"2309.10455","repositories_listed":0,"syntology":null},{"url":null,"slug":"posterior-sampling-algorithms-for","title":"Posterior sampling algorithms for unsupervised speech enhancement with recurrent variational autoencoder","date":"2023-09-19","arxiv_id":"2309.10439","repositories_listed":0,"syntology":null},{"url":null,"slug":"refining-dnn-based-mask-estimation-using-cgmm","title":"Refining DNN-based Mask Estimation using CGMM-based EM Algorithm for Multi-channel Noise Reduction","date":"2023-09-18","arxiv_id":"2309.09630","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-modeling-of-the-denoising-process","title":"Continuous Modeling of the Denoising Process for Speech Enhancement Based on Deep Learning","date":"2023-09-17","arxiv_id":"2309.09270","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifying-robustness-and-fidelity-a","title":"Unifying Robustness and Fidelity: A Comprehensive Study of Pretrained Generative Methods for Speech Enhancement in Adverse Conditions","date":"2023-09-16","arxiv_id":"2309.09028","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-step-knowledge-distillation-for-tiny","title":"Two-Step Knowledge Distillation for Tiny Speech Enhancement","date":"2023-09-15","arxiv_id":"2309.08144","repositories_listed":0,"syntology":null},{"url":null,"slug":"av2wav-diffusion-based-re-synthesis-from","title":"AV2Wav: Diffusion-Based Re-synthesis from Continuous Self-supervised Features for Audio-Visual Speech Enhancement","date":"2023-09-14","arxiv_id":"2309.08030","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-generalization-gap-of-learning","title":"Assessing the Generalization Gap of Learning-Based Speech Enhancement Systems in Noisy and Reverberant Environments","date":"2023-09-12","arxiv_id":"2309.06183","repositories_listed":0,"syntology":null},{"url":"/paper/cleanunet-2-a-hybrid-speech-denoising-model","slug":"cleanunet-2-a-hybrid-speech-denoising-model","title":"CleanUNet 2: A Hybrid Speech Denoising Model on Waveform and Spectrogram","date":"2023-09-12","arxiv_id":"2309.05975","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-signal-based-dccrn-with-overlapped","title":"Causal Signal-Based DCCRN with Overlapped-Frame Prediction for Online Speech Enhancement","date":"2023-09-07","arxiv_id":"2309.03684","repositories_listed":0,"syntology":null},{"url":null,"slug":"spiking-structured-state-space-model-for","title":"Spiking Structured State Space Model for Monaural Speech Enhancement","date":"2023-09-07","arxiv_id":"2309.03641","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-channel-speech-enhancement-with-deep","title":"Single-Channel Speech Enhancement with Deep Complex U-Networks and Probabilistic Latent Space Models","date":"2023-09-04","arxiv_id":"2309.01535","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-robust-speech-emotion-recognition-with","title":"Noise robust speech emotion recognition with signal-to-noise ratio adapting speech enhancement","date":"2023-09-03","arxiv_id":"2309.01164","repositories_listed":0,"syntology":null}],"record_sha256":"44d082eb635635679d73cd18e0b63aebe3df7c0138c3ce7ebd7c4933695fd722","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}