{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speaker-identification/papers/2","list_of":"/task/speaker-identification","task":"Speaker Identification","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":3,"rows_per_page":100,"rows":[101,200],"of":248,"counts":{"archive_papers_tagged":248,"with_a_code_link":74,"where_syntology_ran_a_sample":15,"not_listed_spam_title":0,"listed":248,"listed_where_code_ran":15,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":11,"every_run_a_failure_of_syntologys_instrument":4,"listed_with_a_run_with_no_instrument_failure":11,"listed_every_run_a_failure_of_syntologys_instrument":4,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speaker-identification","prev":"/task/speaker-identification","next":"/task/speaker-identification/papers/3","papers":[{"url":null,"slug":"neural-networks-hear-you-loud-and-clear","title":"Hearing-Loss Compensation Using Deep Neural Networks: A Framework and Results From a Listening Test","date":"2024-03-15","arxiv_id":"2403.10420","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-closer-look-at-wav2vec2-embeddings-for-on","title":"A Closer Look at Wav2Vec2 Embeddings for On-Device Single-Channel Speech Enhancement","date":"2024-03-03","arxiv_id":"2403.01369","repositories_listed":0,"syntology":null},{"url":null,"slug":"unraveling-adversarial-examples-against","title":"Unraveling Adversarial Examples against Speaker Identification -- Techniques for Attack Detection and Victim Model Classification","date":"2024-02-29","arxiv_id":"2402.19355","repositories_listed":0,"syntology":null},{"url":null,"slug":"effect-of-utterance-duration-and-phonetic","title":"Effect of utterance duration and phonetic content on speaker identification using second-order statistical methods","date":"2024-02-26","arxiv_id":"2402.16429","repositories_listed":0,"syntology":null},{"url":null,"slug":"significance-of-chirp-mfcc-as-a-feature-in","title":"Significance of Chirp MFCC as a Feature in Speech and Audio Applications","date":"2024-02-19","arxiv_id":"2402.12239","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-self-supervised-learning-models-with","title":"Probing Self-supervised Learning Models with Target Speech Extraction","date":"2024-02-17","arxiv_id":"2402.13200","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-rhythm-based-speaker-embeddings","title":"Speech Rhythm-Based Speaker Embeddings Extraction from Phonemes and Phoneme Duration for Multi-Speaker Speech Synthesis","date":"2024-02-11","arxiv_id":"2402.07085","repositories_listed":0,"syntology":null},{"url":null,"slug":"post-training-embedding-alignment-for","title":"Post-Training Embedding Alignment for Decoupling Enrollment and Runtime Speaker Recognition Models","date":"2024-01-23","arxiv_id":"2401.12440","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxceleb-esp-preliminary-experiments","title":"Voxceleb-ESP: preliminary experiments detecting Spanish celebrities from their voices","date":"2023-12-20","arxiv_id":"2401.09441","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficiency-oriented-approaches-for-self","title":"Efficiency-oriented approaches for self-supervised speech representation learning","date":"2023-12-18","arxiv_id":"2312.11142","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-representation-learning-1","title":"Privacy-preserving Representation Learning for Speech Understanding","date":"2023-10-26","arxiv_id":"2310.17194","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-accent-dialect-identification-and","title":"Advanced accent/dialect identification and accentedness assessment with multi-embedding models and automatic speech recognition","date":"2023-10-17","arxiv_id":"2310.11004","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-multichannel-speaker-attributed","title":"End-to-end Multichannel Speaker-Attributed ASR: Speaker Guided Decoder and Input Feature Analysis","date":"2023-10-16","arxiv_id":"2310.10106","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-time-training-for-speech","title":"Test-Time Training for Speech","date":"2023-09-19","arxiv_id":"2309.10930","repositories_listed":0,"syntology":null},{"url":null,"slug":"spiking-leaf-a-learnable-auditory-front-end","title":"Spiking-LEAF: A Learnable Auditory front-end for Spiking Neural Networks","date":"2023-09-18","arxiv_id":"2309.09469","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-self-supervised-learning-of","title":"Understanding Self-Supervised Learning of Speech Representation via Invariance and Redundancy Reduction","date":"2023-09-07","arxiv_id":"2309.03619","repositories_listed":0,"syntology":null},{"url":null,"slug":"read-look-or-listen-what-s-needed-for-solving","title":"Read, Look or Listen? What's Needed for Solving a Multimodal Dataset","date":"2023-07-06","arxiv_id":"2307.04532","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxwatch-an-open-set-speaker-recognition","title":"VoxWatch: An open-set speaker recognition benchmark on VoxCeleb","date":"2023-06-30","arxiv_id":"2307.00169","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-specific-thresholding-for-robust","title":"Meta-Learning Framework for End-to-End Imposter Identification in Unseen Speaker Recognition","date":"2023-06-01","arxiv_id":"2306.00952","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-speaker-identification-using-1","title":"Few-Shot Speaker Identification Using Lightweight Prototypical Network with Feature Grouping and Interaction","date":"2023-05-31","arxiv_id":"2305.19541","repositories_listed":0,"syntology":null},{"url":null,"slug":"ordered-and-binary-speaker-embedding","title":"Ordered and Binary Speaker Embedding","date":"2023-05-25","arxiv_id":"2305.16043","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-transferability-of-whisper-based","title":"On the Transferability of Whisper-based Representations for \"In-the-Wild\" Cross-Task Downstream Speech Applications","date":"2023-05-23","arxiv_id":"2305.14546","repositories_listed":0,"syntology":null},{"url":null,"slug":"security-and-privacy-problems-in-voice","title":"Security and Privacy Problems in Voice Assistant Applications: A Survey","date":"2023-04-19","arxiv_id":"2304.09486","repositories_listed":0,"syntology":null},{"url":null,"slug":"hissnet-sound-event-detection-and-speaker","title":"HiSSNet: Sound Event Detection and Speaker Identification via Hierarchical Prototypical Networks for Low-Resource Headphones","date":"2023-03-13","arxiv_id":"2303.07538","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-knowledge-distillation-of-self","title":"Ensemble knowledge distillation of self-supervised speech models","date":"2023-02-24","arxiv_id":"2302.12757","repositories_listed":0,"syntology":null},{"url":null,"slug":"exarn-self-attending-rnn-for-target-speaker","title":"ExARN: self-attending RNN for target speaker extraction","date":"2022-12-02","arxiv_id":"2212.01106","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-label-training-for-text-independent","title":"Multi-Label Training for Text-Independent Speaker Identification","date":"2022-11-14","arxiv_id":"2211.07373","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-utility-balanced-voice-de","title":"Privacy-Utility Balanced Voice De-Identification Using Adversarial Examples","date":"2022-11-10","arxiv_id":"2211.05446","repositories_listed":0,"syntology":null},{"url":null,"slug":"symmetric-saliency-based-adversarial-attack","title":"Symmetric Saliency-based Adversarial Attack To Speaker Identification","date":"2022-10-30","arxiv_id":"2210.16777","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantitative-evidence-on-overlooked-aspects","title":"Quantitative Evidence on Overlooked Aspects of Enrollment Speaker Embeddings for Target Speaker Separation","date":"2022-10-23","arxiv_id":"2210.12635","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-identification-from-emotional-and","title":"Speaker Identification from emotional and noisy speech data using learned voice segregation and Speech VGG","date":"2022-10-23","arxiv_id":"2210.12701","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-independent-speaker-identification","title":"Text Independent Speaker Identification System for Access Control","date":"2022-09-26","arxiv_id":"2209.14335","repositories_listed":0,"syntology":null},{"url":null,"slug":"computing-with-hypervectors-for-efficient","title":"Computing with Hypervectors for Efficient Speaker Identification","date":"2022-08-28","arxiv_id":"2208.13285","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-versus-wide-an-analysis-of-student","title":"Deep versus Wide: An Analysis of Student Architectures for Task-Agnostic Knowledge Distillation of Self-Supervised Speech Models","date":"2022-07-14","arxiv_id":"2207.06867","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-based-multi-view-fusion-and-local","title":"Graph-based Multi-View Fusion and Local Adaptation: Mitigating Within-Household Confusability for Speaker Identification","date":"2022-07-08","arxiv_id":"2207.04081","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-diarization-and-identification-from","title":"Speaker Diarization and Identification from Single-Channel Classroom Audio Recording Using Virtual Microphones","date":"2022-07-01","arxiv_id":"2207.00660","repositories_listed":0,"syntology":null},{"url":null,"slug":"iemotts-toward-robust-cross-speaker-emotion","title":"iEmoTTS: Toward Robust Cross-Speaker Emotion Transfer and Control for Speech Synthesis based on Disentanglement between Prosody and Timbre","date":"2022-06-29","arxiv_id":"2206.14866","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-source-speakers-for-voice","title":"Identifying Source Speakers for Voice Conversion based Spoofing Attacks on Speaker Verification Systems","date":"2022-06-18","arxiv_id":"2206.09103","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategies-to-improve-robustness-of-target","title":"Strategies to Improve Robustness of Target Speech Extraction to Enrollment Variations","date":"2022-06-16","arxiv_id":"2206.08174","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-identification-using-speech","title":"Speaker Identification using Speech Recognition","date":"2022-05-29","arxiv_id":"2205.14649","repositories_listed":0,"syntology":null},{"url":null,"slug":"silence-is-sweeter-than-speech-self","title":"Silence is Sweeter Than Speech: Self-Supervised Model Using Silence to Store Speaker Information","date":"2022-05-08","arxiv_id":"2205.03759","repositories_listed":0,"syntology":null},{"url":null,"slug":"vfhq-a-high-quality-dataset-and-benchmark-for","title":"VFHQ: A High-Quality Dataset and Benchmark for Video Face Super-Resolution","date":"2022-05-06","arxiv_id":"2205.03409","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-speaker-identification-using","title":"Few-Shot Speaker Identification Using Depthwise Separable Convolutional Network with Channel Attention","date":"2022-04-24","arxiv_id":"2204.11180","repositories_listed":0,"syntology":null},{"url":null,"slug":"wabert-a-low-resource-end-to-end-model-for","title":"WaBERT: A Low-resource End-to-end Model for Spoken Language Understanding and Speech-to-BERT Alignment","date":"2022-04-22","arxiv_id":"2204.10461","repositories_listed":0,"syntology":null},{"url":null,"slug":"listen-only-to-me-how-well-can-target-speech","title":"Listen only to me! How well can target speech extraction handle false alarms?","date":"2022-04-11","arxiv_id":"2204.04811","repositories_listed":0,"syntology":null},{"url":null,"slug":"advest-adversarial-perturbation-estimation-to","title":"AdvEst: Adversarial Perturbation Estimation to Classify and Detect Adversarial Attacks against Speaker Identification","date":"2022-04-08","arxiv_id":"2204.03848","repositories_listed":0,"syntology":null},{"url":null,"slug":"karaoker-alignment-free-singing-voice","title":"Karaoker: Alignment-free singing voice synthesis with speech training data","date":"2022-04-08","arxiv_id":"2204.04127","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-relation-networks-for-end-to-end","title":"Improved Relation Networks for End-to-End Speaker Verification and Identification","date":"2022-03-31","arxiv_id":"2203.17218","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuragen-a-low-resource-neural-network-based","title":"NeuraGen-A Low-Resource Neural Network based approach for Gender Classification","date":"2022-03-29","arxiv_id":"2203.15253","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-identification-experiments-under","title":"Speaker Identification Experiments Under Gender De-Identification","date":"2022-03-09","arxiv_id":"2203.04638","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-relevance-of-bandwidth-extension-for","title":"On the relevance of bandwidth extension for speaker identification","date":"2022-02-24","arxiv_id":"2202.13865","repositories_listed":0,"syntology":null},{"url":null,"slug":"openfeat-improving-speaker-identification-by","title":"openFEAT: Improving Speaker Identification by Open-set Few-shot Embedding Adaptation with Transformer","date":"2022-02-24","arxiv_id":"2202.12349","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-watermarking-a-solution-for","title":"Speech watermarking: an approach for the forensic analysis of digital telephonic recordings","date":"2022-02-23","arxiv_id":"2203.02275","repositories_listed":0,"syntology":null},{"url":null,"slug":"pipe-overflow-smashing-voice-authentication","title":"Tubes Among Us: Analog Attack on Automatic Speaker Identification","date":"2022-02-06","arxiv_id":"2202.02751","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-lingual-speaker-identification-from","title":"Cross-Lingual Speaker Identification from Weak Local Evidence","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-exploitation-of-multiple-feature","title":"The exploitation of Multiple Feature Extraction Techniques for Speaker Identification in Emotional States under Disguised Voices","date":"2021-12-15","arxiv_id":"2112.07940","repositories_listed":0,"syntology":null},{"url":null,"slug":"blind-extraction-of-target-speech-source","title":"Target Speech Extraction: Independent Vector Extraction Guided by Supervised Speaker Identification","date":"2021-11-05","arxiv_id":"2111.03482","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-acoustic-features-in-arabic","title":"A Study of Acoustic Features in Arabic Speaker Identification under Noisy Environmental Conditions","date":"2021-10-23","arxiv_id":"2110.12304","repositories_listed":0,"syntology":null},{"url":null,"slug":"power-efficient-analog-features-for-audio","title":"PEAF: Learnable Power Efficient Analog Acoustic Features for Audio Recognition","date":"2021-10-07","arxiv_id":"2110.03715","repositories_listed":0,"syntology":null},{"url":null,"slug":"transcribe-to-diarize-neural-speaker","title":"Transcribe-to-Diarize: Neural Speaker Diarization for Unlimited Number of Speakers using End-to-End Speaker-Attributed ASR","date":"2021-10-07","arxiv_id":"2110.03151","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speaker-identification-for-shared","title":"Improving Speaker Identification for Shared Devices by Adapting Embeddings to Speaker Subsets","date":"2021-09-06","arxiv_id":"2109.02576","repositories_listed":0,"syntology":null},{"url":null,"slug":"qasr-qcri-aljazeera-speech-resource-a-large-1","title":"QASR: QCRI Aljazeera Speech Resource A Large Scale Annotated Arabic Speech Corpus","date":"2021-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-real-time-speaker-diarization-system-based","title":"A Real-time Speaker Diarization System Based on Spatial Spectrum","date":"2021-07-20","arxiv_id":"2107.09321","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-learning-to-classify-and","title":"Representation Learning to Classify and Detect Adversarial Attacks against Speaker and Speech Recognition Systems","date":"2021-07-09","arxiv_id":"2107.04448","repositories_listed":0,"syntology":null},{"url":null,"slug":"qasr-qcri-aljazeera-speech-resource-a-large","title":"QASR: QCRI Aljazeera Speech Resource -- A Large Scale Annotated Arabic Speech Corpus","date":"2021-06-24","arxiv_id":"2106.13000","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusion-of-embeddings-networks-for-robust","title":"Fusion of Embeddings Networks for Robust Combination of Text Dependent and Independent Speaker Recognition","date":"2021-06-18","arxiv_id":"2106.10169","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-based-label-propagation-for-semi","title":"Graph-based Label Propagation for Semi-Supervised Speaker Identification","date":"2021-06-15","arxiv_id":"2106.08207","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-diarization-for-variable-number-of","title":"End-to-End Diarization for Variable Number of Speakers with Local-Global Networks and Discriminative Speaker Embeddings","date":"2021-05-05","arxiv_id":"2105.02096","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speaker-attributed-asr-with","title":"End-to-End Speaker-Attributed ASR with Transformer","date":"2021-04-05","arxiv_id":"2104.02128","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-multi-talker-speech-recognition","title":"Streaming Multi-talker Speech Recognition with Joint Speaker Identification","date":"2021-04-05","arxiv_id":"2104.02109","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-paralinguistics-in-tamil-speech","title":"A Survey on Paralinguistics in Tamil Speech Processing","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-privacy-with-smart-digital-assistants","title":"Voice Privacy with Smart Digital Assistants in Educational Settings","date":"2021-03-24","arxiv_id":"2104.11038","repositories_listed":0,"syntology":null},{"url":null,"slug":"triplet-loss-based-embeddings-for-forensic","title":"Triplet loss based embeddings for forensic speaker identification in Spanish","date":"2021-02-24","arxiv_id":"2102.12564","repositories_listed":0,"syntology":null},{"url":null,"slug":"casa-based-speaker-identification-using","title":"CASA-Based Speaker Identification Using Cascaded GMM-CNN Classifier in Noisy and Emotional Talking Conditions","date":"2021-02-11","arxiv_id":"2102.05894","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-attribution-with-voice-profiles-by","title":"Speaker attribution with voice profiles by graph-based semi-supervised learning","date":"2021-02-06","arxiv_id":"2102.03634","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypothesis-stitcher-for-end-to-end-speaker","title":"Hypothesis Stitcher for End-to-End Speaker-attributed ASR on Long-form Multi-talker Recordings","date":"2021-01-06","arxiv_id":"2101.01853","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-few-shot-audio-classification","title":"A Study of Few-Shot Audio Classification","date":"2020-12-02","arxiv_id":"2012.01573","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-far-are-we-from-robust-voice-conversion-a","title":"How Far Are We from Robust Voice Conversion: A Survey","date":"2020-11-24","arxiv_id":"2011.12063","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-emotion-detection-with-transfer","title":"Multi-Modal Emotion Detection with Transfer Learning","date":"2020-11-13","arxiv_id":"2011.07065","repositories_listed":0,"syntology":null},{"url":null,"slug":"t-vectors-weakly-supervised-speaker","title":"T-vectors: Weakly Supervised Speaker Identification Using Hierarchical Transformer Model","date":"2020-10-29","arxiv_id":"2010.16071","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lightweight-speaker-recognition-system","title":"A Lightweight Speaker Recognition System Using Timbre Properties","date":"2020-10-12","arxiv_id":"2010.05502","repositories_listed":0,"syntology":null},{"url":null,"slug":"remarks-on-optimal-scores-for-speaker","title":"Remarks on Optimal Scores for Speaker Recognition","date":"2020-10-10","arxiv_id":"2010.04862","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-faults-in-our-asrs-an-overview-of-attacks","title":"SoK: The Faults in our ASRs: An Overview of Attacks against Automatic Speech Recognition and Speaker Identification Systems","date":"2020-07-13","arxiv_id":"2007.06622","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-speaker-counting-speech-recognition-and","title":"Joint Speaker Counting, Speech Recognition, and Speaker Identification for Overlapped Speech of Any Number of Speakers","date":"2020-06-19","arxiv_id":"2006.10930","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrated-replay-spoofing-aware-text","title":"Integrated Replay Spoofing-aware Text-independent Speaker Verification","date":"2020-06-10","arxiv_id":"2006.05599","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-and-posture-classification-using","title":"Speaker and Posture Classification using Instantaneous Intraspeech Breathing Features","date":"2020-05-25","arxiv_id":"2005.12230","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-training-of-hierarchical","title":"Weakly Supervised Training of Hierarchical Attention Networks for Speaker Identification","date":"2020-05-15","arxiv_id":"2005.07817","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-recognition-in-bengali-language-from","title":"Speaker Recognition in Bengali Language from Nonlinear Features","date":"2020-04-15","arxiv_id":"2004.07820","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-recurrent-denoising-autoencoder","title":"End-to-end Recurrent Denoising Autoencoder Embeddings for Speaker Identification","date":"2020-03-13","arxiv_id":"2003.07688","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-neural-networks-for-automatic-speech","title":"Deep Neural Networks for Automatic Speech Processing: A Survey from Large Corpora to Limited Data","date":"2020-03-09","arxiv_id":"2003.04241","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-identification-using-eeg","title":"Speaker Identification using EEG","date":"2020-03-07","arxiv_id":"2003.04733","repositories_listed":0,"syntology":null},{"url":"/paper/multi-task-learning-network-for-emotion","slug":"multi-task-learning-network-for-emotion","title":"Multi-Task Learning with Auxiliary Speaker Identification for Conversational Emotion Recognition","date":"2020-03-03","arxiv_id":"2003.01478","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-enhancement-using-self-adaptation-and","title":"Speech Enhancement using Self-Adaptation and Multi-Head Self-Attention","date":"2020-02-14","arxiv_id":"2002.05873","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-speaker-recognition-using-speech","title":"Robust Speaker Recognition Using Speech Enhancement And Attention Model","date":"2020-01-14","arxiv_id":"2001.05031","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-speaker-embedding-de-mixing-in-two","title":"Supervised Speaker Embedding De-Mixing in Two-Speaker Environment","date":"2020-01-14","arxiv_id":"2001.06397","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-deterministic-plus-stochastic-model-of","title":"The Deterministic plus Stochastic Model of the Residual Signal and its Applications","date":"2019-12-29","arxiv_id":"2001.01000","repositories_listed":0,"syntology":null},{"url":null,"slug":"advances-in-online-audio-visual-meeting","title":"Advances in Online Audio-Visual Meeting Transcription","date":"2019-12-10","arxiv_id":"1912.04979","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-adversarial-representation","title":"Privacy-Preserving Adversarial Representation Learning in ASR: Reality or Illusion?","date":"2019-11-12","arxiv_id":"1911.04913","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-initialization-of-lstm-networks","title":"Supervised Initialization of LSTM Networks for Fundamental Frequency Detection in Noisy Speech Signals","date":"2019-11-11","arxiv_id":"1911.04580","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-audio-membership-inference-attack","title":"Reducing audio membership inference attack accuracy to chance: 4 defenses","date":"2019-10-31","arxiv_id":"1911.01888","repositories_listed":0,"syntology":null}],"record_sha256":"9964835ffee12eb37cea0dd6b6a1c8e3d5190e281c78a666911171dfea5312b0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}