{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-enhancement/papers/4","list_of":"/task/speech-enhancement","task":"Speech Enhancement","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":10,"rows_per_page":100,"rows":[301,400],"of":982,"counts":{"archive_papers_tagged":982,"with_a_code_link":280,"where_syntology_ran_a_sample":54,"not_listed_spam_title":0,"listed":982,"listed_where_code_ran":54,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":45,"every_run_a_failure_of_syntologys_instrument":9,"listed_with_a_run_with_no_instrument_failure":45,"listed_every_run_a_failure_of_syntologys_instrument":9,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-enhancement","prev":"/task/speech-enhancement/papers/3","next":"/task/speech-enhancement/papers/5","papers":[{"url":null,"slug":"a-semantic-information-based-hierarchical","title":"A Semantic Information-based Hierarchical Speech Enhancement Method Using Factorized Codec and Diffusion Model","date":"2025-05-20","arxiv_id":"2505.13843","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-noise-robustness-of-llm-based-zero","title":"Improving Noise Robustness of LLM-based Zero-shot TTS via Discrete Acoustic Token Denoising","date":"2025-05-20","arxiv_id":"2505.13830","repositories_listed":0,"syntology":null},{"url":null,"slug":"mddm-a-multi-view-discriminative-enhanced","title":"MDDM: A Multi-view Discriminative Enhanced Diffusion-based Model for Speech Enhancement","date":"2025-05-19","arxiv_id":"2505.13029","repositories_listed":0,"syntology":null},{"url":null,"slug":"rovo-robust-voice-protection-against","title":"RoVo: Robust Voice Protection Against Unauthorized Speech Synthesis with Embedding-Level Perturbations","date":"2025-05-19","arxiv_id":"2505.12686","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-architecture-and-unsupervised-speech","title":"Unified Architecture and Unsupervised Speech Disentanglement for Speaker Embedding-Free Enrollment in Personalized Speech Enhancement","date":"2025-05-18","arxiv_id":"2505.12288","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-deep-learning-for-complex-speech","title":"A Survey of Deep Learning for Complex Speech Spectrograms","date":"2025-05-13","arxiv_id":"2505.08694","repositories_listed":0,"syntology":null},{"url":null,"slug":"normalize-everything-a-preconditioned","title":"Normalize Everything: A Preconditioned Magnitude-Preserving Architecture for Diffusion-Based Speech Enhancement","date":"2025-05-08","arxiv_id":"2505.05216","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-speech-recognition-with-schrodinger","title":"Robust Speech Recognition with Schrödinger Bridge-Based Speech Enhancement","date":"2025-05-07","arxiv_id":"2505.04237","repositories_listed":0,"syntology":null},{"url":null,"slug":"swinlip-an-efficient-visual-speech-encoder","title":"SwinLip: An Efficient Visual Speech Encoder for Lip Reading Using Swin Transformer","date":"2025-05-07","arxiv_id":"2505.04394","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-much-to-dereverberate-low-latency-single","title":"How much to Dereverberate? Low-Latency Single-Channel Speech Enhancement in Distant Microphone Scenarios","date":"2025-05-02","arxiv_id":"2505.01338","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-speech-intelligibility-in-older","title":"Predicting speech intelligibility in older adults using the Gammachirp Envelope Similarity Index, GESI","date":"2025-04-20","arxiv_id":"2504.14437","repositories_listed":0,"syntology":null},{"url":null,"slug":"ditse-high-fidelity-generative-speech","title":"DiTSE: High-Fidelity Generative Speech Enhancement via Latent Diffusion Transformers","date":"2025-04-13","arxiv_id":"2504.09381","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-filter-bank-based-neural-method-for","title":"Spatial-Filter-Bank-Based Neural Method for Multichannel Speech Enhancement","date":"2025-04-02","arxiv_id":"2504.01392","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-low-power-streaming-speech-enhancement","title":"A Low-Power Streaming Speech Enhancement Accelerator For Edge Devices","date":"2025-03-27","arxiv_id":"2503.21335","repositories_listed":0,"syntology":null},{"url":null,"slug":"magnitude-phase-dual-path-speech-enhancement","title":"Magnitude-Phase Dual-Path Speech Enhancement Network based on Self-Supervised Embedding and Perceptual Contrast Stretch Boosting","date":"2025-03-27","arxiv_id":"2503.21571","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-spectrogram-separation-and-tdoa","title":"Joint Spectrogram Separation and TDOA Estimation using Optimal Transport","date":"2025-03-24","arxiv_id":"2503.18600","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-speech-production-model-for-radar","title":"A Speech Production Model for Radar: Connecting Speech Acoustics with Radar-Measured Vibrations","date":"2025-03-19","arxiv_id":"2503.15627","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-autoencoder-for-personalized","title":"Variational Autoencoder for Personalized Pathological Speech Enhancement","date":"2025-03-18","arxiv_id":"2503.14036","repositories_listed":0,"syntology":null},{"url":null,"slug":"linguistic-knowledge-transfer-learning-for","title":"Linguistic Knowledge Transfer Learning for Speech Enhancement","date":"2025-03-10","arxiv_id":"2503.07078","repositories_listed":0,"syntology":null},{"url":null,"slug":"prose-diffusion-priors-for-speech-enhancement","title":"ProSE: Diffusion Priors for Speech Enhancement","date":"2025-03-09","arxiv_id":"2503.06375","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-speech-quality-through-the","title":"Enhancing Speech Quality through the Integration of BGRU and Transformer Architectures","date":"2025-02-25","arxiv_id":"2502.17911","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-enhancement-using-continuous","title":"Speech Enhancement Using Continuous Embeddings of Neural Audio Codec","date":"2025-02-22","arxiv_id":"2502.16240","repositories_listed":0,"syntology":null},{"url":null,"slug":"lmfca-net-a-lightweight-model-for-multi","title":"LMFCA-Net: A Lightweight Model for Multi-Channel Speech Enhancement with Efficient Narrow-Band and Cross-Band Attention","date":"2025-02-17","arxiv_id":"2502.11462","repositories_listed":0,"syntology":null},{"url":null,"slug":"taps-throat-and-acoustic-paired-speech","title":"TAPS: Throat and Acoustic Paired Speech Dataset for Deep Learning-Based Speech Enhancement","date":"2025-02-17","arxiv_id":"2502.11478","repositories_listed":0,"syntology":null},{"url":null,"slug":"microphone-array-geometry-independent-multi","title":"Microphone Array Geometry Independent Multi-Talker Distant ASR: NTT System for the DASR Task of the CHiME-8 Challenge","date":"2025-02-14","arxiv_id":"2502.09859","repositories_listed":0,"syntology":null},{"url":null,"slug":"advances-in-microphone-array-processing-and","title":"Advances in Microphone Array Processing and Multichannel Speech Enhancement","date":"2025-02-13","arxiv_id":"2502.09037","repositories_listed":0,"syntology":null},{"url":null,"slug":"gense-generative-speech-enhancement-via","title":"GenSE: Generative Speech Enhancement via Language Models using Hierarchical Modeling","date":"2025-02-05","arxiv_id":"2502.02942","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-multi-modality-gaps-of-audio","title":"Bridging The Multi-Modality Gaps of Audio, Visual and Linguistic for Speech Enhancement","date":"2025-01-23","arxiv_id":"2501.13375","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-data-augmentation-challenge-zero","title":"Generative Data Augmentation Challenge: Zero-Shot Speech Synthesis for Personalized Speech Enhancement","date":"2025-01-23","arxiv_id":"2501.13372","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-a-posteriori-speech-presence","title":"Learning-based A Posteriori Speech Presence Probability Estimation and Applications","date":"2025-01-23","arxiv_id":"2501.13642","repositories_listed":0,"syntology":null},{"url":"/paper/let-ssms-be-convnets-state-space-modeling","slug":"let-ssms-be-convnets-state-space-modeling","title":"Let SSMs be ConvNets: State-space Modeling with Optimal Tensor Contractions","date":"2025-01-22","arxiv_id":"2501.13230","repositories_listed":0,"syntology":null},{"url":null,"slug":"up-cycle-senet-unpaired-phase-aware-speech","title":"UP-Cycle-SENet: Unpaired Phase-aware Speech Enhancement Using Deep Complex Cycle Adversarial Networks","date":"2025-01-22","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dfingernet-noise-adaptive-speech-enhancement","title":"DFingerNet: Noise-Adaptive Speech Enhancement for Hearing Aids","date":"2025-01-17","arxiv_id":"2501.10525","repositories_listed":0,"syntology":null},{"url":null,"slug":"microphone-array-signal-processing-and-deep","title":"Microphone Array Signal Processing and Deep Learning for Speech Enhancement","date":"2025-01-13","arxiv_id":"2501.07215","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-speech-enhancement-with-limited","title":"Multi-modal Speech Enhancement with Limited Electromyography Channels","date":"2025-01-11","arxiv_id":"2501.06530","repositories_listed":0,"syntology":null},{"url":null,"slug":"artifact-free-sound-quality-in-dnn-based","title":"Artifact-free Sound Quality in DNN-based Closed-loop Systems for Audio Processing","date":"2025-01-07","arxiv_id":"2501.04116","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-speech-enhancement-with-predicting","title":"Causal Speech Enhancement with Predicting Semantics based on Quantized Self-supervised Learning Features","date":"2024-12-26","arxiv_id":"2412.19248","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-directed-speech-enhancement-with-dual","title":"Neural Directed Speech Enhancement with Dual Microphone Array in High Noise Scenario","date":"2024-12-24","arxiv_id":"2412.18141","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-on-the-potential-of-kan-in","title":"From KAN to GR-KAN: Advancing Speech Enhancement with KAN-Based Methodology","date":"2024-12-23","arxiv_id":"2412.17778","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-speech-enhancement-with-dynamic","title":"Scalable Speech Enhancement with Dynamic Channel Pruning","date":"2024-12-22","arxiv_id":"2412.17121","repositories_listed":0,"syntology":null},{"url":null,"slug":"scale-this-not-that-investigating-key-dataset","title":"Scale This, Not That: Investigating Key Dataset Attributes for Efficient Speech Enhancement Scaling","date":"2024-12-19","arxiv_id":"2412.14890","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-effects-of-diffusion-based","title":"Investigating the Effects of Diffusion-based Conditional Generative Speech Models Used for Speech Enhancement on Dysarthric Speech","date":"2024-12-18","arxiv_id":"2412.13933","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-impact-of-discriminative-and","title":"Evaluating the Impact of Discriminative and Generative E2E Speech Enhancement Models on Syllable Stress Preservation","date":"2024-12-11","arxiv_id":"2412.08306","repositories_listed":0,"syntology":null},{"url":null,"slug":"touchtts-an-embarrassingly-simple-tts","title":"TouchTTS: An Embarrassingly Simple TTS Framework that Everyone Can Touch","date":"2024-12-11","arxiv_id":"2412.08237","repositories_listed":0,"syntology":null},{"url":null,"slug":"salmonn-omni-a-codec-free-llm-for-full-duplex","title":"SALMONN-omni: A Codec-free LLM for Full-duplex Speech Understanding and Generation","date":"2024-11-27","arxiv_id":"2411.18138","repositories_listed":0,"syntology":null},{"url":null,"slug":"ghostrnn-reducing-state-redundancy-in-rnn","title":"GhostRNN: Reducing State Redundancy in RNN with Cheap Operations","date":"2024-11-20","arxiv_id":"2411.14489","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-advanced-speech-signal-processing-a","title":"Towards Advanced Speech Signal Processing: A Statistical Perspective on Convolution-Based Architectures and its Applications","date":"2024-11-20","arxiv_id":"2411.18636","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-neural-denoising-vocoder-for-clean-waveform","title":"A Neural Denoising Vocoder for Clean Waveform Generation from Noisy Mel-Spectrogram based on Amplitude and Phase Predictions","date":"2024-11-19","arxiv_id":"2411.12268","repositories_listed":0,"syntology":null},{"url":null,"slug":"sav-se-scene-aware-audio-visual-speech","title":"SAV-SE: Scene-aware Audio-Visual Speech Enhancement with Selective State Space Model","date":"2024-11-12","arxiv_id":"2411.07751","repositories_listed":0,"syntology":null},{"url":null,"slug":"dcf-ds-deep-cascade-fusion-of-diarization-and","title":"DCF-DS: Deep Cascade Fusion of Diarization and Separation for Speech Recognition under Realistic Single-Channel Conditions","date":"2024-11-11","arxiv_id":"2411.06667","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-state-space-model-for-monaural","title":"Selective State Space Model for Monaural Speech Enhancement","date":"2024-11-09","arxiv_id":"2411.06217","repositories_listed":0,"syntology":null},{"url":null,"slug":"modulating-state-space-model-with-slowfast","title":"Modulating State Space Model with SlowFast Framework for Compute-Efficient Ultra Low-Latency Speech Enhancement","date":"2024-11-04","arxiv_id":"2411.02019","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-aware-unified-source-separation","title":"Task-Aware Unified Source Separation","date":"2024-10-31","arxiv_id":"2410.23987","repositories_listed":0,"syntology":null},{"url":null,"slug":"run-time-adaptation-of-neural-beamforming-for","title":"Run-Time Adaptation of Neural Beamforming for Robust Speech Dereverberation and Denoising","date":"2024-10-30","arxiv_id":"2410.22805","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-diarization-and-separation-of","title":"Simultaneous Diarization and Separation of Meetings through the Integration of Statistical Mixture Models","date":"2024-10-28","arxiv_id":"2410.21455","repositories_listed":0,"syntology":null},{"url":null,"slug":"elaichi-enhancing-low-resource-tts-by","title":"ELAICHI: Enhancing Low-resource TTS by Addressing Infrequent and Low-frequency Character Bigrams","date":"2024-10-23","arxiv_id":"2410.17901","repositories_listed":0,"syntology":null},{"url":null,"slug":"gan-based-speech-enhancement-for-low-snr","title":"GAN-Based Speech Enhancement for Low SNR Using Latent Feature Conditioning","date":"2024-10-17","arxiv_id":"2410.13599","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-rlhf-to-align-speech-enhancement","title":"Using RLHF to align speech enhancement approaches to mean-opinion quality scores","date":"2024-10-17","arxiv_id":"2410.13182","repositories_listed":0,"syntology":null},{"url":null,"slug":"finally-fast-and-universal-speech-enhancement","title":"FINALLY: fast and universal speech enhancement with studio-like quality","date":"2024-10-08","arxiv_id":"2410.05920","repositories_listed":0,"syntology":null},{"url":null,"slug":"relunet-relative-channel-fusion-u-net-for","title":"RelUNet: Relative Channel Fusion U-Net for Multichannel Speech Enhancement","date":"2024-10-07","arxiv_id":"2410.05019","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-based-unsupervised-audio-visual","title":"Diffusion-based Unsupervised Audio-visual Speech Enhancement","date":"2024-10-04","arxiv_id":"2410.05301","repositories_listed":0,"syntology":null},{"url":null,"slug":"restorative-speech-enhancement-a-progressive","title":"Restorative Speech Enhancement: A Progressive Approach Using SE and Codec Modules","date":"2024-10-02","arxiv_id":"2410.01150","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-clustering-techniques-for-speech","title":"Advanced Clustering Techniques for Speech Signal Enhancement: A Review and Metanalysis of Fuzzy C-Means, K-Means, and Kernel Fuzzy C-Means Methods","date":"2024-09-28","arxiv_id":"2409.19448","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-boosting-low-latency-live-speech","title":"Speech Boosting: Low-Latency Live Speech Enhancement for TWS Earbuds","date":"2024-09-27","arxiv_id":"2409.18705","repositories_listed":0,"syntology":null},{"url":null,"slug":"mc-semamba-a-simple-multi-channel-extension","title":"MC-SEMamba: A Simple Multi-channel Extension of SEMamba","date":"2024-09-26","arxiv_id":"2409.17898","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-sub-millisecond-latency-real-time","title":"Towards Sub-millisecond Latency Real-Time Speech Enhancement Models on Hearables","date":"2024-09-26","arxiv_id":"2409.18239","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-explicit-consistency-preserving-loss","title":"An Explicit Consistency-Preserving Loss Function for Phase Reconstruction and Speech Enhancement","date":"2024-09-24","arxiv_id":"2409.16282","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-audio-visual-speech-enhancement","title":"Robust Audio-Visual Speech Enhancement: Correcting Misassignments in Complex Environments with Advanced Post-Processing","date":"2024-09-22","arxiv_id":"2409.14554","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-audio-visual-soundscape","title":"Self-Supervised Audio-Visual Soundscape Stylization","date":"2024-09-22","arxiv_id":"2409.14340","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometry-constrained-eeg-channel-selection","title":"Geometry-Constrained EEG Channel Selection for Brain-Assisted Speech Enhancement","date":"2024-09-19","arxiv_id":"2409.12520","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-declipping-transformer-with-complex","title":"Speech-Declipping Transformer with Complex Spectrogram and Learnerble Temporal Features","date":"2024-09-19","arxiv_id":"2409.12416","repositories_listed":0,"syntology":null},{"url":"/paper/dense-tsnet-dense-connected-two-stage","slug":"dense-tsnet-dense-connected-two-stage","title":"Dense-TSNet: Dense Connected Two-Stage Structure for Ultra-Lightweight Speech Enhancement","date":"2024-09-18","arxiv_id":"2409.11725","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-joint-spectral-and-spatial","title":"Leveraging Joint Spectral and Spatial Learning with MAMBA for Multichannel Speech Enhancement","date":"2024-09-16","arxiv_id":"2409.10376","repositories_listed":0,"syntology":null},{"url":null,"slug":"tcg-crest-system-description-for-the-second","title":"TCG CREST System Description for the Second DISPLACE Challenge","date":"2024-09-16","arxiv_id":"2409.15356","repositories_listed":0,"syntology":null},{"url":null,"slug":"ultra-low-latency-speech-enhancement-a","title":"Ultra-Low Latency Speech Enhancement - A Comprehensive Study","date":"2024-09-16","arxiv_id":"2409.10358","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-mamba-in-speech-processing-by-self","title":"Rethinking Mamba in Speech Processing by Self-Supervised Models","date":"2024-09-11","arxiv_id":"2409.07273","repositories_listed":0,"syntology":null},{"url":null,"slug":"dewinder-single-channel-wind-noise-reduction","title":"DeWinder: Single-Channel Wind Noise Reduction using Ultrasound Sensing","date":"2024-09-10","arxiv_id":"2409.06137","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-based-speech-enhancement-with","title":"Diffusion-based Speech Enhancement with Schrödinger Bridge and Symmetric Noise Schedule","date":"2024-09-08","arxiv_id":"2409.05116","repositories_listed":0,"syntology":null},{"url":null,"slug":"tf-mamba-a-time-frequency-network-for-sound","title":"TF-Mamba: A Time-Frequency Network for Sound Source Localization","date":"2024-09-08","arxiv_id":"2409.05034","repositories_listed":0,"syntology":null},{"url":"/paper/raw-speech-enhancement-with-deep-state-space","slug":"raw-speech-enhancement-with-deep-state-space","title":"aTENNuate: Optimized Real-time Speech Enhancement with Deep SSMs on Raw Audio","date":"2024-09-05","arxiv_id":"2409.03377","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-residual-extraction-based-pre","title":"Progressive Residual Extraction based Pre-training for Speech Representation Learning","date":"2024-08-31","arxiv_id":"2409.00387","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-masking-with-explicit-time-context","title":"Spectral Masking with Explicit Time-Context Windowing for Neural Network-Based Monaural Speech Enhancement","date":"2024-08-28","arxiv_id":"2408.15582","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-gated-recurrent-neural-network-for","title":"Dynamic Gated Recurrent Neural Network for Compute-efficient Speech Enhancement","date":"2024-08-22","arxiv_id":"2408.12425","repositories_listed":0,"syntology":null},{"url":null,"slug":"dpsnn-spiking-neural-network-for-low-latency","title":"DPSNN: Spiking Neural Network for Low-Latency Streaming Speech Enhancement","date":"2024-08-14","arxiv_id":"2408.07388","repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogeneous-space-fusion-and-dual-dimension","title":"Heterogeneous Space Fusion and Dual-Dimension Attention: A New Paradigm for Speech Enhancement","date":"2024-08-13","arxiv_id":"2408.06911","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-shot-distributed-node-specific-signal","title":"One-Shot Distributed Node-Specific Signal Estimation with Non-Overlapping Latent Subspaces in Acoustic Sensor Networks","date":"2024-08-07","arxiv_id":"2408.03752","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctpulse-close-talk-and-pseudo-label-based-far","title":"ctPuLSE: Close-Talk, and Pseudo-Label Based Far-Field, Speech Enhancement","date":"2024-07-28","arxiv_id":"2407.19485","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-bandwidth-expansion-via-high-fidelity","title":"Speech Bandwidth Expansion Via High Fidelity Generative Adversarial Networks","date":"2024-07-26","arxiv_id":"2407.18571","repositories_listed":0,"syntology":null},{"url":"/paper/schrodinger-bridge-for-generative-speech","slug":"schrodinger-bridge-for-generative-speech","title":"Schrödinger Bridge for Generative Speech Enhancement","date":"2024-07-22","arxiv_id":"2407.16074","repositories_listed":0,"syntology":null},{"url":null,"slug":"rt-la-voce-real-time-low-snr-audio-visual","title":"RT-LA-VocE: Real-Time Low-SNR Audio-Visual Speech Enhancement","date":"2024-07-10","arxiv_id":"2407.07825","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-face-mask-speech-enhancement","title":"Unsupervised Face-Masked Speech Enhancement Using Generative Adversarial Networks With Human-in-the-Loop Assessment Metrics","date":"2024-07-02","arxiv_id":"2407.01939","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-source-conversational-ai-with","title":"Open-Source Conversational AI with SpeechBrain 1.0","date":"2024-06-29","arxiv_id":"2407.00463","repositories_listed":0,"syntology":null},{"url":null,"slug":"dasb-discrete-audio-and-speech-benchmark","title":"DASB -- Discrete Audio and Speech Benchmark","date":"2024-06-20","arxiv_id":"2406.14294","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-based-generative-modeling-with","title":"Diffusion-based Generative Modeling with Discriminative Guidance for Streamable Speech Enhancement","date":"2024-06-19","arxiv_id":"2406.13471","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-exploration-of-length-generalization-in","title":"An Exploration of Length Generalization in Transformer-Based Speech Enhancement","date":"2024-06-17","arxiv_id":"2406.11401","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatially-constrained-vs-unconstrained","title":"Spatially constrained vs. unconstrained filtering in neural spatiospectral filters for multichannel speech enhancement","date":"2024-06-17","arxiv_id":"2406.11376","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-speech-enhancement-without-a","title":"Personalized Speech Enhancement Without a Separate Speaker Embedding Model","date":"2024-06-14","arxiv_id":"2406.09928","repositories_listed":0,"syntology":null},{"url":null,"slug":"flowavse-efficient-audio-visual-speech","title":"FlowAVSE: Efficient Audio-Visual Speech Enhancement with Conditional Flow Matching","date":"2024-06-13","arxiv_id":"2406.09286","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparative-analysis-of-personalized-voice","title":"Comparative Analysis of Personalized Voice Activity Detection Systems: Assessing Real-World Effectiveness","date":"2024-06-12","arxiv_id":"2406.09443","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-feature-guided-diffusion-model","title":"Pre-training Feature Guided Diffusion Model for Speech Enhancement","date":"2024-06-11","arxiv_id":"2406.07646","repositories_listed":0,"syntology":null}],"record_sha256":"db19226cf57ec6e0e2b05b95402fd8004ecf90e0e9fc7c1e02269423c506fd7a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}