{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-enhancement/papers/6","list_of":"/task/speech-enhancement","task":"Speech Enhancement","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":10,"rows_per_page":100,"rows":[501,600],"of":982,"counts":{"archive_papers_tagged":982,"with_a_code_link":280,"where_syntology_ran_a_sample":54,"not_listed_spam_title":0,"listed":982,"listed_where_code_ran":54,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":45,"every_run_a_failure_of_syntologys_instrument":9,"listed_with_a_run_with_no_instrument_failure":45,"listed_every_run_a_failure_of_syntologys_instrument":9,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-enhancement","prev":"/task/speech-enhancement/papers/5","next":"/task/speech-enhancement/papers/7","papers":[{"url":null,"slug":"rep2wav-noise-robust-text-to-speech-using","title":"Rep2wav: Noise Robust text-to-speech Using self-supervised representations","date":"2023-08-28","arxiv_id":"2308.14553","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-time-frequency-conformers-for","title":"Exploiting Time-Frequency Conformers for Music Audio Enhancement","date":"2023-08-24","arxiv_id":"2308.12599","repositories_listed":0,"syntology":null},{"url":null,"slug":"adverb-visually-guided-audio-dereverberation","title":"AdVerb: Visually Guided Audio Dereverberation","date":"2023-08-23","arxiv_id":"2308.12370","repositories_listed":0,"syntology":null},{"url":null,"slug":"convoifilter-a-case-study-of-doing-cocktail","title":"Convoifilter: A case study of doing cocktail party speech recognition","date":"2023-08-22","arxiv_id":"2308.11380","repositories_listed":0,"syntology":null},{"url":null,"slug":"speechx-neural-codec-language-model-as-a","title":"SpeechX: Neural Codec Language Model as a Versatile Speech Transformer","date":"2023-08-14","arxiv_id":"2308.06873","repositories_listed":0,"syntology":null},{"url":null,"slug":"target-speech-extraction-with-conditional","title":"Target Speech Extraction with Conditional Diffusion Model","date":"2023-08-08","arxiv_id":"2308.03987","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-monaural-speech-enhancement-using","title":"Efficient Monaural Speech Enhancement using Spectrum Attention Fusion","date":"2023-08-04","arxiv_id":"2308.02263","repositories_listed":0,"syntology":null},{"url":null,"slug":"samba-speech-enhancement-with-asynchronous-ad","title":"SAMbA: Speech enhancement with Asynchronous ad-hoc Microphone Arrays","date":"2023-07-31","arxiv_id":"2307.16582","repositories_listed":0,"syntology":null},{"url":null,"slug":"pcnn-a-lightweight-parallel-conformer-neural","title":"PCNN: A Lightweight Parallel Conformer Neural Network for Efficient Monaural Speech Enhancement","date":"2023-07-28","arxiv_id":"2307.15251","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-intrusive-intelligibility-predictor-for","title":"Non Intrusive Intelligibility Predictor for Hearing Impaired Individuals using Self Supervised Speech Representations","date":"2023-07-25","arxiv_id":"2307.13423","repositories_listed":0,"syntology":null},{"url":null,"slug":"slmgan-exploiting-speech-language-model","title":"SLMGAN: Exploiting Speech Language Model Representations for Unsupervised Zero-Shot Voice Conversion in GANs","date":"2023-07-18","arxiv_id":"2307.09435","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-bit-rate-binaural-link-for-improved-ultra","title":"Low bit rate binaural link for improved ultra low-latency low-complexity multichannel speech enhancement in Hearing Aids","date":"2023-07-17","arxiv_id":"2307.08858","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-enhancement-using-self","title":"Audio-Visual Speech Enhancement Using Self-supervised Learning to Improve Speech Intelligibility in Cochlear Implant Simulations","date":"2023-07-15","arxiv_id":"2307.07748","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-end-to-end-multi-channel-speech","title":"Audio-visual End-to-end Multi-channel Speech Separation, Dereverberation and Recognition","date":"2023-07-06","arxiv_id":"2307.02909","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-loss-convolutional-network-with-time","title":"Multi-Loss Convolutional Network with Time-Frequency Attention for Speech Enhancement","date":"2023-06-15","arxiv_id":"2306.08956","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-normalization-for-fine-tuning-self","title":"Feature Normalization for Fine-tuning Self-Supervised Models in Speech Enhancement","date":"2023-06-14","arxiv_id":"2306.08406","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-speech-enhancement-with-deep","title":"Unsupervised speech enhancement with deep dynamical generative speech and noise models","date":"2023-06-13","arxiv_id":"2306.07820","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-enhancement-with","title":"Audio-Visual Speech Enhancement With Selective Off-Screen Speech Extraction","date":"2023-06-10","arxiv_id":"2306.06495","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-encoder-decoder-and-dual-path","title":"Efficient Encoder-Decoder and Dual-Path Conformer for Comprehensive Feature Learning in Speech Enhancement","date":"2023-06-09","arxiv_id":"2306.05861","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-stage-autoencoder-neural-network-for-3d","title":"Convolutional Recurrent Neural Network with Attention for 3D Speech Enhancement","date":"2023-06-08","arxiv_id":"2306.04987","repositories_listed":0,"syntology":null},{"url":null,"slug":"effcrn-an-efficient-convolutional-recurrent","title":"EffCRN: An Efficient Convolutional Recurrent Network for High-Performance Speech Enhancement","date":"2023-06-05","arxiv_id":"2306.02778","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-behavior-of-intrusive-and-non","title":"On the Behavior of Intrusive and Non-intrusive Speech Enhancement Metrics in Predictive and Generative Settings","date":"2023-06-05","arxiv_id":"2306.03014","repositories_listed":0,"syntology":null},{"url":null,"slug":"influence-of-lossy-speech-codecs-on-hearing","title":"Influence of Lossy Speech Codecs on Hearing-aid, Binaural Sound Source Localisation using DNNs","date":"2023-06-04","arxiv_id":"2306.02344","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-enhancement-with-score","title":"Audio-Visual Speech Enhancement with Score-Based Generative Models","date":"2023-06-02","arxiv_id":"2306.01432","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-crowdsourcing-design-with-comparison","title":"On Crowdsourcing-design with Comparison Category Rating for Evaluating Speech Enhancement Algorithms","date":"2023-06-02","arxiv_id":"2306.01538","repositories_listed":0,"syntology":null},{"url":null,"slug":"harmonic-enhancement-using-learnable-comb","title":"Harmonic enhancement using learnable comb filter for light-weight full-band speech enhancement model","date":"2023-06-01","arxiv_id":"2306.00812","repositories_listed":0,"syntology":null},{"url":null,"slug":"downstream-task-agnostic-speech-enhancement","title":"Downstream Task Agnostic Speech Enhancement with Self-Supervised Representation Loss","date":"2023-05-24","arxiv_id":"2305.14723","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-ultrasound-tongue-images-for","title":"Incorporating Ultrasound Tongue Images for Audio-Visual Speech Enhancement through Knowledge Distillation","date":"2023-05-24","arxiv_id":"2305.14933","repositories_listed":0,"syntology":null},{"url":null,"slug":"se-bridge-speech-enhancement-with-consistent","title":"SE-Bridge: Speech Enhancement with Consistent Brownian Bridge","date":"2023-05-23","arxiv_id":"2305.13796","repositories_listed":0,"syntology":null},{"url":null,"slug":"dccrn-kws-an-audio-bias-based-model-for-noise","title":"DCCRN-KWS: an audio bias based model for noise robust small-footprint keyword spotting","date":"2023-05-21","arxiv_id":"2305.12331","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-based-speech-enhancement-with-joint","title":"Diffusion-Based Speech Enhancement with Joint Generative and Predictive Decoders","date":"2023-05-18","arxiv_id":"2305.10734","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-based-signal-refiner-for-speech","title":"Diffusion-based Signal Refiner for Speech Separation","date":"2023-05-10","arxiv_id":"2305.05857","repositories_listed":0,"syntology":null},{"url":null,"slug":"all-information-is-necessary-integrating","title":"All Information is Necessary: Integrating Speech Positive and Negative Information by Contrastive Learning for Speech Enhancement","date":"2023-04-26","arxiv_id":"2304.13439","repositories_listed":0,"syntology":null},{"url":null,"slug":"array-configuration-agnostic-personal-voice","title":"Array Configuration-Agnostic Personal Voice Activity Detection Based on Spatial Coherence","date":"2023-04-18","arxiv_id":"2304.08887","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-speech-enhancement-with-very-low","title":"Neural Speech Enhancement with Very Low Algorithmic Latency and Complexity via Integrated Full- and Sub-Band Modeling","date":"2023-04-18","arxiv_id":"2304.08707","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-future-of-hearing-aid-technology","title":"The future of hearing aid technology","date":"2023-04-13","arxiv_id":"2304.06786","repositories_listed":0,"syntology":null},{"url":null,"slug":"wav2code-restore-clean-speech-representations","title":"Wav2code: Restore Clean Speech Representations via Codebook Lookup for Noise-Robust ASR","date":"2023-04-11","arxiv_id":"2304.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-speech-enhancement-using","title":"Attention-based Speech Enhancement Using Human Quality Perception Modelling","date":"2023-03-23","arxiv_id":"2303.13685","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-diffusion-model-for-speech-synthesis-a","title":"A Survey on Audio Diffusion Models: Text To Speech Synthesis and Enhancement in Generative AI","date":"2023-03-23","arxiv_id":"2303.13336","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-in-speech-processing-a-survey","title":"Transformers in Speech Processing: A Survey","date":"2023-03-21","arxiv_id":"2303.11607","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-perceptual-quality-intelligibility","title":"Improving Perceptual Quality, Intelligibility, and Acoustics on VoIP Platforms","date":"2023-03-16","arxiv_id":"2303.09048","repositories_listed":0,"syntology":null},{"url":null,"slug":"subspace-hybrid-beamforming-for-head-worn","title":"Subspace Hybrid Beamforming for Head-worn Microphone Arrays","date":"2023-03-15","arxiv_id":"2303.08967","repositories_listed":0,"syntology":null},{"url":null,"slug":"localizing-spatial-information-in-neural","title":"Localizing Spatial Information in Neural Spatiospectral Filters","date":"2023-03-14","arxiv_id":"2303.08052","repositories_listed":0,"syntology":null},{"url":null,"slug":"tea-pse-3-0-tencent-ethereal-audio-lab","title":"TEA-PSE 3.0: Tencent-Ethereal-Audio-Lab Personalized Speech Enhancement System For ICASSP 2023 DNS Challenge","date":"2023-03-14","arxiv_id":"2303.07704","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-dereverberation-noise-and-interferer","title":"Adaptive Dereverberation, Noise and Interferer Reduction Using Sparse Weighted Linearly Constrained Minimum Power Beamforming","date":"2023-03-13","arxiv_id":"2303.07027","repositories_listed":0,"syntology":null},{"url":null,"slug":"blind-acoustic-room-parameter-estimation","title":"Blind Acoustic Room Parameter Estimation Using Phase Features","date":"2023-03-13","arxiv_id":"2303.07449","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-speech-enhancement-network","title":"Guided Speech Enhancement Network","date":"2023-03-13","arxiv_id":"2303.07486","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-audio-visual-end-to-end-speech","title":"Real-Time Audio-Visual End-to-End Speech Enhancement","date":"2023-03-13","arxiv_id":"2303.07005","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-intent-classification-accuracy","title":"Improving the Intent Classification accuracy in Noisy Environment","date":"2023-03-12","arxiv_id":"2303.06585","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-modeling-with-a-hierarchical","title":"Speech Modeling with a Hierarchical Transformer Dynamical VAE","date":"2023-03-07","arxiv_id":"2303.09404","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-audio-video-enhancement-with-a","title":"Real-time Audio Video Enhancement \\\\with a Microphone Array and Headphones","date":"2023-03-02","arxiv_id":"2303.00949","repositories_listed":0,"syntology":null},{"url":null,"slug":"dfsnet-a-steerable-neural-beamformer","title":"DFSNet: A Steerable Neural Beamformer Invariant to Microphone Array Configuration for Real-Time, Low-Latency Speech Enhancement","date":"2023-02-26","arxiv_id":"2302.13407","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-variance-aware-real-time-speech","title":"Time-Variance Aware Real-Time Speech Enhancement","date":"2023-02-25","arxiv_id":"2302.13063","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-for-unified-real-time","title":"A Framework for Unified Real-time Personalized and Non-Personalized Speech Enhancement","date":"2023-02-23","arxiv_id":"2302.11768","repositories_listed":0,"syntology":null},{"url":null,"slug":"metric-oriented-speech-enhancement-using","title":"Metric-oriented Speech Enhancement using Diffusion Probabilistic Model","date":"2023-02-23","arxiv_id":"2302.11989","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-noise-adaptation-using-data","title":"Unsupervised Noise adaptation using Data Simulation","date":"2023-02-23","arxiv_id":"2302.11981","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-speech-enhancement-with-dynamic","title":"Real-time speech enhancement with dynamic attention span","date":"2023-02-21","arxiv_id":"2302.10377","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-speech-enhancement-combining","title":"Personalized speech enhancement combining band-split RNN and speaker attentive module","date":"2023-02-20","arxiv_id":"2302.09953","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-speech-enhancement-using-spectral","title":"Real-Time Speech Enhancement Using Spectral Subtraction with Minimum Statistics and Spectral Floor","date":"2023-02-20","arxiv_id":"2302.10313","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-enhancement-with-multi-granularity","title":"Speech Enhancement with Multi-granularity Vector Quantization","date":"2023-02-16","arxiv_id":"2302.08342","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-channel-target-speaker-extraction-with","title":"Multi-Channel Target Speaker Extraction with Refinement: The WavLab Submission to the Second Clarity Enhancement Challenge","date":"2023-02-15","arxiv_id":"2302.07928","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-batching-variable-size-inputs-for-training","title":"On Batching Variable Size Inputs for Training End-to-End Speech Enhancement Systems","date":"2023-01-25","arxiv_id":"2301.10587","repositories_listed":0,"syntology":null},{"url":null,"slug":"cellular-network-speech-enhancement-removing","title":"Cellular Network Speech Enhancement: Removing Background and Transmission Noise","date":"2023-01-22","arxiv_id":"2301.09027","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceive-and-predict-self-supervised-speech","title":"Perceive and predict: self-supervised speech representation based loss functions for speech enhancement","date":"2023-01-11","arxiv_id":"2301.04388","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-complex-valued-deep-neural","title":"Rethinking complex-valued deep neural networks for monaural speech enhancement","date":"2023-01-11","arxiv_id":"2301.04320","repositories_listed":0,"syntology":null},{"url":"/paper/revise-self-supervised-speech-resynthesis","slug":"revise-self-supervised-speech-resynthesis","title":"ReVISE: Self-Supervised Speech Resynthesis with Visual Input for Universal and Generalized Speech Enhancement","date":"2022-12-21","arxiv_id":"2212.11377","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-estimation-in-deep-speech","title":"Uncertainty Estimation in Deep Speech Enhancement Using Complex Gaussian Mixture Models","date":"2022-12-09","arxiv_id":"2212.04831","repositories_listed":0,"syntology":null},{"url":null,"slug":"selector-enhancer-learning-dynamic-selection","title":"Selector-Enhancer: Learning Dynamic Selection of Local and Non-local Attention Operation for Speech Enhancement","date":"2022-12-07","arxiv_id":"2212.03408","repositories_listed":0,"syntology":null},{"url":null,"slug":"injecting-spatial-information-for-monaural","title":"Injecting Spatial Information for Monaural Speech Enhancement via Knowledge Distillation","date":"2022-12-02","arxiv_id":"2212.01012","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-neural-network-techniques-for-monaural","title":"Deep neural network techniques for monaural speech enhancement: state of the art analysis","date":"2022-12-01","arxiv_id":"2212.00369","repositories_listed":0,"syntology":null},{"url":null,"slug":"stereo-speech-enhancement-using-custom-mid","title":"Stereo Speech Enhancement Using Custom Mid-Side Signals and Monaural Processing","date":"2022-11-25","arxiv_id":"2211.14378","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-acoustic-compensation-and-adaptive","title":"Dynamic Acoustic Compensation and Adaptive Focal Training for Personalized Speech Enhancement","date":"2022-11-22","arxiv_id":"2211.12097","repositories_listed":0,"syntology":null},{"url":"/paper/d2net-a-denoising-and-dereverberation-network","slug":"d2net-a-denoising-and-dereverberation-network","title":"D²Net: A Denoising and Dereverberation Network Based on Two-branch Encoder and Dual-path Transformer","date":"2022-11-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"la-voce-low-snr-audio-visual-speech","title":"LA-VocE: Low-SNR Audio-visual Speech Enhancement using Neural Vocoders","date":"2022-11-20","arxiv_id":"2211.10999","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-wavlm-on-speech-enhancement","title":"Exploring WavLM on Speech Enhancement","date":"2022-11-18","arxiv_id":"2211.09988","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-two-stage-deep-representation-learning","title":"A Two-Stage Deep Representation Learning-Based Speech Enhancement Method Using Variational Autoencoder and Adversarial Training","date":"2022-11-16","arxiv_id":"2211.09166","repositories_listed":0,"syntology":null},{"url":null,"slug":"array-configuration-agnostic-personalized","title":"Array Configuration-Agnostic Personalized Speech Enhancement using Long-Short-Term Spatial Coherence","date":"2022-11-16","arxiv_id":"2211.08748","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-heteroscedastic-uncertainty-in","title":"Leveraging Heteroscedastic Uncertainty in Learning Complex Spectral Mapping for Single-channel Speech Enhancement","date":"2022-11-16","arxiv_id":"2211.08624","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-label-training-for-text-independent","title":"Multi-Label Training for Text-Independent Speaker Identification","date":"2022-11-14","arxiv_id":"2211.07373","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-potential-of-neural-speech-synthesis","title":"The Potential of Neural Speech Synthesis-based Data Augmentation for Personalized Speech Enhancement","date":"2022-11-14","arxiv_id":"2211.07493","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-attention-is-all-you-need-real-time","title":"Cross-Attention is all you need: Real-Time Streaming Transformers for Personalised Speech Enhancement","date":"2022-11-08","arxiv_id":"2211.04346","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffphase-generative-diffusion-based-stft","title":"DiffPhase: Generative Diffusion-based STFT Phase Retrieval","date":"2022-11-08","arxiv_id":"2211.04332","repositories_listed":0,"syntology":null},{"url":null,"slug":"egocentric-audio-visual-noise-suppression","title":"Egocentric Audio-Visual Noise Suppression","date":"2022-11-07","arxiv_id":"2211.03643","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-the-trade-off-in-personalized-speech","title":"Breaking the trade-off in personalized speech enhancement with cross-task knowledge distillation","date":"2022-11-05","arxiv_id":"2211.02944","repositories_listed":0,"syntology":null},{"url":null,"slug":"cold-diffusion-for-speech-enhancement","title":"Cold Diffusion for Speech Enhancement","date":"2022-11-04","arxiv_id":"2211.02527","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-joint-personalized-speech","title":"Real-Time Joint Personalized Speech Enhancement and Acoustic Echo Cancellation","date":"2022-11-04","arxiv_id":"2211.02773","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-enhancement-using-ego-noise-references","title":"Speech enhancement using ego-noise references with a microphone array embedded in an unmanned aerial vehicle","date":"2022-11-04","arxiv_id":"2211.02690","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-kernels-and-channel-attention-with","title":"Dynamic Kernels and Channel Attention for Low Resource Speaker Verification","date":"2022-11-03","arxiv_id":"2211.02000","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-autoregression-a-novel-trick-to","title":"Iterative autoregression: a novel trick to improve your low-latency speech enhancement model","date":"2022-11-03","arxiv_id":"2211.01751","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-noisy-target-training-for-dnn","title":"Analysis of Noisy-target Training for DNN-based speech enhancement","date":"2022-11-02","arxiv_id":"2211.01198","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-enhancement-with-a-deep","title":"Audio-visual speech enhancement with a deep Kalman filter generative model","date":"2022-11-02","arxiv_id":"2211.00988","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-and-efficient-speech-enhancement-with","title":"Fast and efficient speech enhancement with variational autoencoders","date":"2022-11-02","arxiv_id":"2211.02728","repositories_listed":0,"syntology":null},{"url":null,"slug":"weighted-variance-variational-autoencoder-for","title":"A weighted-variance variational autoencoder model for speech enhancement","date":"2022-11-02","arxiv_id":"2211.00990","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-preliminary-study-of-the-application-of","title":"A Preliminary Study of the Application of Discrete Wavelet Transform Features in Conv-TasNet Speech Enhancement Model","date":"2022-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-the-compressed-spectral-loss-for","title":"Exploiting the compressed spectral loss for the learning of the DEMUCS speech enhancement network","date":"2022-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sca-streaming-cross-attention-alignment-for","title":"SCA: Streaming Cross-attention Alignment for Echo Cancellation","date":"2022-11-01","arxiv_id":"2211.00589","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-enhancement-and","title":"Audio-Visual Speech Enhancement and Separation by Utilizing Multi-Modal Self-Supervised Embeddings","date":"2022-10-31","arxiv_id":"2210.17456","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-gated-neural-network-with-attention","title":"Parallel Gated Neural Network With Attention Mechanism For Speech Enhancement","date":"2022-10-26","arxiv_id":"2210.14509","repositories_listed":0,"syntology":null},{"url":"/paper/scp-gan-self-correcting-discriminator","slug":"scp-gan-self-correcting-discriminator","title":"SCP-GAN: Self-Correcting Discriminator Optimization for Training Consistency Preserving Metric GAN on Speech Enhancement Tasks","date":"2022-10-26","arxiv_id":"2210.14474","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-frame-structure-for-cloud-based-audio","title":"A Novel Frame Structure for Cloud-Based Audio-Visual Speech Enhancement in Multimodal Hearing-aids","date":"2022-10-24","arxiv_id":"2210.13127","repositories_listed":0,"syntology":null}],"record_sha256":"43c86d43d3f4ab22ddb901b4684009b240bb754c022a40e2e8b3d366373e9465","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}