{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speaker-recognition/papers/3","list_of":"/task/speaker-recognition","task":"Speaker Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":5,"rows_per_page":100,"rows":[201,300],"of":435,"counts":{"archive_papers_tagged":435,"with_a_code_link":102,"where_syntology_ran_a_sample":17,"not_listed_spam_title":0,"listed":435,"listed_where_code_ran":17,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":15,"every_run_a_failure_of_syntologys_instrument":2,"listed_with_a_run_with_no_instrument_failure":15,"listed_every_run_a_failure_of_syntologys_instrument":2,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speaker-recognition","prev":"/task/speaker-recognition/papers/2","next":"/task/speaker-recognition/papers/4","papers":[{"url":null,"slug":"i4u-system-description-for-nist-sre-20-cts","title":"I4U System Description for NIST SRE'20 CTS Challenge","date":"2022-11-02","arxiv_id":"2211.01091","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-end-to-end-speaker-diarization-in-the","title":"Late Audio-Visual Fusion for In-The-Wild Speaker Diarization","date":"2022-11-02","arxiv_id":"2211.01299","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-representation-learning-for-1","title":"Disentangled representation learning for multilingual speaker recognition","date":"2022-11-01","arxiv_id":"2211.00437","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-speaker-recognition-encoders-for","title":"Universal speaker recognition encoders for different speech segments duration","date":"2022-10-28","arxiv_id":"2210.16231","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-training-of-speaker-encoder","title":"Self-Supervised Training of Speaker Encoder with Multi-Modal Diverse Positive Pairs","date":"2022-10-27","arxiv_id":"2210.15385","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speech-representation-learning-via","title":"Improving Speech Representation Learning via Speech-level and Phoneme-level Masking Approach","date":"2022-10-25","arxiv_id":"2210.13805","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-learning-of-generalised","title":"Large-scale learning of generalised representations for speaker recognition","date":"2022-10-20","arxiv_id":"2210.10985","repositories_listed":0,"syntology":null},{"url":null,"slug":"superb-slt-2022-challenge-on-generalization","title":"SUPERB @ SLT 2022: Challenge on Generalization and Efficiency of Self-Supervised Speech Representation Learning","date":"2022-10-16","arxiv_id":"2210.08634","repositories_listed":0,"syntology":null},{"url":null,"slug":"thuee-system-description-for-nist-2020-sre","title":"THUEE system description for NIST 2020 SRE CTS challenge","date":"2022-10-12","arxiv_id":"2210.06111","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-dku-dukeece-diarization-system-for-the","title":"The DKU-DukeECE Diarization System for the VoxCeleb Speaker Recognition Challenge 2022","date":"2022-10-04","arxiv_id":"2210.01677","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-kriston-ai-system-for-the-voxceleb","title":"The Kriston AI System for the VoxCeleb Speaker Recognition Challenge 2022","date":"2022-09-23","arxiv_id":"2209.11433","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-speakin-system-description-for-cnsrc2022","title":"The SpeakIn System Description for CNSRC2022","date":"2022-09-22","arxiv_id":"2209.10846","repositories_listed":0,"syntology":null},{"url":null,"slug":"gist-aiter-system-for-the-diarization-task-of","title":"GIST-AiTeR System for the Diarization Task of the 2022 VoxCeleb Speaker Recognition Challenge","date":"2022-09-21","arxiv_id":"2209.10357","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-returnzero-system-for-voxceleb-speaker","title":"The ReturnZero System for VoxCeleb Speaker Recognition Challenge 2022","date":"2022-09-21","arxiv_id":"2209.10147","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-royalflush-system-for-voxceleb-speaker","title":"The Royalflush System for VoxCeleb Speaker Recognition Challenge 2022","date":"2022-09-19","arxiv_id":"2209.09010","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-benchmark-for-understanding-and-generating","title":"A Benchmark for Understanding and Generating Dialogue between Characters in Stories","date":"2022-09-18","arxiv_id":"2209.08524","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-speaker-representation-learning","title":"Disentangled Speaker Representation Learning via Mutual Information Minimization","date":"2022-08-17","arxiv_id":"2208.08012","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-attention-and-data-independent","title":"Attention and DCT based Global Context Modeling for Text-independent Speaker Recognition","date":"2022-08-04","arxiv_id":"2208.02778","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-aware-attack-creating-adversarial","title":"Perception-Aware Attack: Creating Adversarial Music via Reverse-Engineering Human Perception","date":"2022-07-26","arxiv_id":"2207.13192","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-based-multi-view-fusion-and-local","title":"Graph-based Multi-View Fusion and Local Adaptation: Mitigating Within-Household Confusability for Speaker Identification","date":"2022-07-08","arxiv_id":"2207.04081","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-speaker-representation","title":"A Hierarchical Speaker Representation Framework for One-shot Singing Voice Conversion","date":"2022-06-28","arxiv_id":"2206.13762","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-end-to-end-private-automatic-speaker","title":"Towards End-to-End Private Automatic Speaker Recognition","date":"2022-06-23","arxiv_id":"2206.11750","repositories_listed":0,"syntology":null},{"url":null,"slug":"as2t-arbitrary-source-to-target-adversarial","title":"AS2T: Arbitrary Source-To-Target Adversarial Attack on Speaker Recognition Systems","date":"2022-06-07","arxiv_id":"2206.03351","repositories_listed":0,"syntology":null},{"url":null,"slug":"far-field-speaker-recognition-benchmark","title":"Far-Field Speaker Recognition Benchmark Derived From The DiPCo Corpus","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"wecantalk-a-new-multi-language-multi-modal","title":"WeCanTalk: A New Multi-language, Multi-modal Resource for Speaker Recognition","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-recognition-of-speakers-for-consent","title":"Dynamic Recognition of Speakers for Consent Management by Contrastive Embedding Replay","date":"2022-05-17","arxiv_id":"2205.08459","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-does-self-supervised-learning-for-speech","title":"Why does Self-Supervised Learning for Speech Recognition Benefit Speaker Recognition?","date":"2022-04-27","arxiv_id":"2204.12765","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-convolutional-network-based-semi","title":"Graph Convolutional Network Based Semi-Supervised Learning on Multi-Speaker Meeting Data","date":"2022-04-25","arxiv_id":"2204.11501","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-2021-nist-speaker-recognition-evaluation","title":"The 2021 NIST Speaker Recognition Evaluation","date":"2022-04-21","arxiv_id":"2204.10242","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-nist-cts-speaker-recognition-challenge","title":"The NIST CTS Speaker Recognition Challenge","date":"2022-04-21","arxiv_id":"2204.10228","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangled-speech-representation-learning","title":"Disentangled Speech Representation Learning Based on Factorized Hierarchical Variational Autoencoder with Self-Supervised Objective","date":"2022-04-05","arxiv_id":"2204.02166","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-speaker-recognition-with-transformers","title":"Robust Speaker Recognition with Transformers Using wav2vec 2.0","date":"2022-03-28","arxiv_id":"2203.15095","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-curriculum-learning-for-1","title":"Curriculum learning for self-supervised speaker verification","date":"2022-03-28","arxiv_id":"2203.14525","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-recognition-by-means-of-a-combination","title":"Speaker recognition by means of a combination of linear and nonlinear predictive models","date":"2022-03-07","arxiv_id":"2203.03190","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-relevance-of-language-in-speaker","title":"On the relevance of language in speaker recognition","date":"2022-03-04","arxiv_id":"2203.01992","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-of-several","title":"A comparative study of several parameterizations for speaker recognition","date":"2022-02-24","arxiv_id":"2203.00513","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-recognition-improvement-using-blind","title":"Speaker recognition improvement using blind inversion of distortions","date":"2022-02-23","arxiv_id":"2203.01164","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-of-the-art-in-speaker-recognition","title":"State-of-the-art in speaker recognition","date":"2022-02-23","arxiv_id":"2202.12705","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-emotion-recognition-using-transfer","title":"Multimodal Emotion Recognition using Transfer Learning from Speaker Recognition and BERT-based models","date":"2022-02-16","arxiv_id":"2202.08974","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-speaker-recognition-training","title":"Self-supervised Speaker Recognition Training Using Human-Machine Dialogues","date":"2022-02-07","arxiv_id":"2202.03484","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-coral-algorithm-for-unsupervised-domain-1","title":"The CORAL++ Algorithm for Unsupervised Domain Adaptation of Speaker Recogntion","date":"2022-02-02","arxiv_id":"2202.01092","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-quality-and-pitch-features-in","title":"Voice Quality and Pitch Features in Transformer-Based Speech Recognition","date":"2021-12-21","arxiv_id":"2112.11391","repositories_listed":0,"syntology":null},{"url":null,"slug":"stc-speaker-recognition-systems-for-the-nist","title":"STC speaker recognition systems for the NIST SRE 2021","date":"2021-11-03","arxiv_id":"2111.02298","repositories_listed":0,"syntology":null},{"url":null,"slug":"black-box-adversarial-attacks-on-commercial","title":"Black-box Adversarial Attacks on Commercial Speech Platforms with Minimal Information","date":"2021-10-19","arxiv_id":"2110.09714","repositories_listed":0,"syntology":null},{"url":null,"slug":"tackling-the-score-shift-in-cross-lingual","title":"Tackling the Score Shift in Cross-Lingual Speaker Verification by Exploiting Language Information","date":"2021-10-18","arxiv_id":"2110.09150","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-view-self-attention-based-transformer","title":"Multi-View Self-Attention Based Transformer for Speaker Recognition","date":"2021-10-11","arxiv_id":"2110.05036","repositories_listed":0,"syntology":null},{"url":null,"slug":"assurance-monitoring-of-learning-enabled","title":"Assurance Monitoring of Learning Enabled Cyber-Physical Systems Using Inductive Conformal Prediction based on Distance Learning","date":"2021-10-07","arxiv_id":"2110.03120","repositories_listed":0,"syntology":null},{"url":null,"slug":"north-america-bixby-speaker-diarization","title":"North America Bixby Speaker Diarization System for the VoxCeleb Speaker Recognition Challenge 2021","date":"2021-09-28","arxiv_id":"2109.13518","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-jhu-submission-to-voxsrc-21-track-3","title":"The JHU submission to VoxSRC-21: Track 3","date":"2021-09-28","arxiv_id":"2109.13425","repositories_listed":0,"syntology":null},{"url":null,"slug":"hello-it-s-me-deep-learning-based-speech","title":"\"Hello, It's Me\": Deep Learning-based Speech Synthesis Attacks in the Real World","date":"2021-09-20","arxiv_id":"2109.09598","repositories_listed":0,"syntology":null},{"url":"/paper/multilingual-audio-visual-smartphone-dataset","slug":"multilingual-audio-visual-smartphone-dataset","title":"Multilingual Audio-Visual Smartphone Dataset And Evaluation","date":"2021-09-09","arxiv_id":"2109.04138","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-idlab-voxceleb-speaker-recognition-1","title":"The IDLAB VoxCeleb Speaker Recognition Challenge 2021 System Description","date":"2021-09-09","arxiv_id":"2109.04070","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-dku-dukeece-system-for-the-self","title":"The DKU-DukeECE System for the Self-Supervision Speaker Verification Task of the 2021 VoxCeleb Speaker Recognition Challenge","date":"2021-09-07","arxiv_id":"2109.02853","repositories_listed":0,"syntology":null},{"url":null,"slug":"xmuspeech-system-for-voxceleb-speaker","title":"XMUSPEECH System for VoxCeleb Speaker Recognition Challenge 2021","date":"2021-09-06","arxiv_id":"2109.02549","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-dku-dukeece-lenovo-system-for-the","title":"The DKU-DukeECE-Lenovo System for the Diarization Task of the 2021 VoxCeleb Speaker Recognition Challenge","date":"2021-09-05","arxiv_id":"2109.02002","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-phonexia-voxceleb-speaker-recognition","title":"The Phonexia VoxCeleb Speaker Recognition Challenge 2021 System Description","date":"2021-09-05","arxiv_id":"2109.02052","repositories_listed":0,"syntology":null},{"url":null,"slug":"nist-sre-cts-superset-a-large-scale-dataset","title":"NIST SRE CTS Superset: A large-scale dataset for telephony speaker recognition","date":"2021-08-16","arxiv_id":"2108.07118","repositories_listed":0,"syntology":null},{"url":null,"slug":"xi-vector-embedding-for-speaker-recognition","title":"Xi-Vector Embedding for Speaker Recognition","date":"2021-08-12","arxiv_id":"2108.05679","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-speech-emotion-recognition-using","title":"Improved Speech Emotion Recognition using Transfer Learning and Spectrogram Augmentation","date":"2021-08-05","arxiv_id":"2108.02510","repositories_listed":0,"syntology":null},{"url":null,"slug":"dropout-regularization-for-self-supervised","title":"Dropout Regularization for Self-Supervised Learning of Transformer Encoder Speech Representation","date":"2021-07-09","arxiv_id":"2107.04227","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-learning-to-classify-and","title":"Representation Learning to Classify and Detect Adversarial Attacks against Speaker and Speech Recognition Systems","date":"2021-07-09","arxiv_id":"2107.04448","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-do-end-to-end-speech-models-learn-about","title":"What do End-to-End Speech Models Learn about Speaker, Language and Channel Information? A Layer-wise and Neuron-level Analysis","date":"2021-07-01","arxiv_id":"2107.00439","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusion-of-embeddings-networks-for-robust","title":"Fusion of Embeddings Networks for Robust Combination of Text Dependent and Independent Speaker Recognition","date":"2021-06-18","arxiv_id":"2106.10169","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-based-label-propagation-for-semi","title":"Graph-based Label Propagation for Semi-Supervised Speaker Identification","date":"2021-06-15","arxiv_id":"2106.08207","repositories_listed":0,"syntology":null},{"url":null,"slug":"utterance-partitioning-for-speaker","title":"Utterance partitioning for speaker recognition: an experimental review and analysis with new findings under GMM-SVM framework","date":"2021-05-25","arxiv_id":"2105.11728","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-fairness-in-speaker-recognition","title":"Improving Fairness in Speaker Recognition","date":"2021-04-29","arxiv_id":"2104.14067","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-learning-for-few-shot-audio","title":"Semi Supervised Learning For Few-shot Audio Classification By Episodic Triplet Mining","date":"2021-02-16","arxiv_id":"2102.08074","repositories_listed":0,"syntology":null},{"url":null,"slug":"content-aware-speaker-embeddings-for-speaker","title":"Content-Aware Speaker Embeddings for Speaker Diarisation","date":"2021-02-12","arxiv_id":"2102.06467","repositories_listed":0,"syntology":null},{"url":null,"slug":"study-of-pre-processing-defenses-against","title":"Study of Pre-processing Defenses against Adversarial Attacks on State-of-the-art Speaker Recognition Systems","date":"2021-01-22","arxiv_id":"2101.08909","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxsrc-2020-the-second-voxceleb-speaker","title":"VoxSRC 2020: The Second VoxCeleb Speaker Recognition Challenge","date":"2020-12-12","arxiv_id":"2012.06867","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-recognition-based-on-deep-learning-an","title":"Speaker Recognition Based on Deep Learning: An Overview","date":"2020-12-02","arxiv_id":"2012.00931","repositories_listed":0,"syntology":null},{"url":null,"slug":"synth2aug-cross-domain-speaker-recognition","title":"Synth2Aug: Cross-domain speaker recognition with TTS synthesized speech","date":"2020-11-24","arxiv_id":"2011.11818","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-lstm-based-end-to-end-text","title":"An Empirical Study on Text-Independent Speaker Verification based on the GE2E Method","date":"2020-11-10","arxiv_id":"2011.04896","repositories_listed":0,"syntology":null},{"url":null,"slug":"query-expansion-system-for-the-voxceleb","title":"Query Expansion System for the VoxCeleb Speaker Recognition Challenge 2020","date":"2020-11-04","arxiv_id":"2011.02882","repositories_listed":0,"syntology":null},{"url":null,"slug":"shanerun-system-description-to-voxceleb","title":"ShaneRun System Description to VoxCeleb Speaker Recognition Challenge 2020","date":"2020-11-03","arxiv_id":"2011.01518","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-xx205-system-for-the-voxceleb-speaker","title":"The xx205 System for the VoxCeleb Speaker Recognition Challenge 2020","date":"2020-10-31","arxiv_id":"2011.00200","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-defense-for-deep-speaker","title":"Adversarial defense for deep speaker recognition using hybrid adversarial training","date":"2020-10-30","arxiv_id":"2010.16038","repositories_listed":0,"syntology":null},{"url":null,"slug":"copypaste-an-augmentation-method-for-speech","title":"CopyPaste: An Augmentation Method for Speech Emotion Recognition","date":"2020-10-27","arxiv_id":"2010.14602","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-upc-speaker-verification-system-submitted","title":"The UPC Speaker Verification System Submitted to VoxCeleb Speaker Recognition Challenge 2020 (VoxSRC-20)","date":"2020-10-27","arxiv_id":"2010.10937","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-learning-of-disentangled-speech","title":"Unsupervised Learning of Disentangled Speech Content and Style Representation","date":"2020-10-24","arxiv_id":"2010.12973","repositories_listed":0,"syntology":null},{"url":null,"slug":"momentum-contrast-speaker-representation","title":"Momentum Contrast Speaker Representation Learning","date":"2020-10-22","arxiv_id":"2010.11457","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-huawei-speaker-diarisation-system-for-the","title":"The HUAWEI Speaker Diarisation System for the VoxCeleb Speaker Diarisation Challenge","date":"2020-10-22","arxiv_id":"2010.11657","repositories_listed":0,"syntology":null},{"url":null,"slug":"tongji-university-undergraduate-team-for-the","title":"Tongji University Undergraduate Team for the VoxCeleb Speaker Recognition Challenge2020","date":"2020-10-20","arxiv_id":"2010.10145","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lightweight-speaker-recognition-system","title":"A Lightweight Speaker Recognition System Using Timbre Properties","date":"2020-10-12","arxiv_id":"2010.05502","repositories_listed":0,"syntology":null},{"url":null,"slug":"remarks-on-optimal-scores-for-speaker","title":"Remarks on Optimal Scores for Speaker Recognition","date":"2020-10-10","arxiv_id":"2010.04862","repositories_listed":0,"syntology":null},{"url":null,"slug":"clova-baseline-system-for-the-voxceleb","title":"Clova Baseline System for the VoxCeleb Speaker Recognition Challenge 2020","date":"2020-09-29","arxiv_id":"2009.14153","repositories_listed":0,"syntology":null},{"url":null,"slug":"siamese-capsule-network-for-end-to-end","title":"Siamese Capsule Network for End-to-End Speaker Recognition In The Wild","date":"2020-09-28","arxiv_id":"2009.13480","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowing-what-to-listen-to-early-attention-for","title":"Fine-grained Early Frequency Attention for Deep Speaker Representation Learning","date":"2020-09-03","arxiv_id":"2009.01822","repositories_listed":0,"syntology":null},{"url":null,"slug":"estimating-uniqueness-of-human-voice-usingi","title":"Estimating Uniqueness of I-Vector Representation of Human Voice","date":"2020-08-27","arxiv_id":"2008.11985","repositories_listed":0,"syntology":null},{"url":null,"slug":"length-and-noise-aware-training-techniques","title":"Length- and Noise-aware Training Techniques for Short-utterance Speaker Recognition","date":"2020-08-27","arxiv_id":"2008.12218","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepvox-discovering-features-from-raw-audio","title":"DeepVOX: Discovering Features from Raw Audio for Speaker Recognition in Non-ideal Audio Signals","date":"2020-08-26","arxiv_id":"2008.11668","repositories_listed":0,"syntology":null},{"url":null,"slug":"they-are-wearing-a-mask-identification-of","title":"They are wearing a mask! Identification of Subjects Wearing a Surgical Mask from their Speech by means of x-vectors and Fisher Vectors","date":"2020-08-23","arxiv_id":"2008.10014","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepsonar-towards-effective-and-robust","title":"DeepSonar: Towards Effective and Robust Detection of AI-Synthesized Fake Voices","date":"2020-08-15","arxiv_id":"2005.13770","repositories_listed":0,"syntology":null},{"url":null,"slug":"compact-speaker-embedding-lrx-vector","title":"Compact Speaker Embedding: lrx-vector","date":"2020-08-11","arxiv_id":"2008.05011","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speaker-recognition-with-a-cross","title":"Audio-visual Speaker Recognition with a Cross-modal Discriminative Network","date":"2020-08-10","arxiv_id":"2008.03894","repositories_listed":0,"syntology":null},{"url":null,"slug":"jukebox-a-multilingual-singer-recognition","title":"JukeBox: A Multilingual Singer Recognition Dataset","date":"2020-08-08","arxiv_id":"2008.03507","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-machine-of-few-words-interactive-speaker","title":"A Machine of Few Words -- Interactive Speaker Recognition with Reinforcement Learning","date":"2020-08-07","arxiv_id":"2008.03127","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-on-device-speaker-verification","title":"Improving on-device speaker verification using federated learning with privacy","date":"2020-08-06","arxiv_id":"2008.02651","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-attention-encoding-and-pooling-for","title":"Self-attention encoding and pooling for speaker recognition","date":"2020-08-03","arxiv_id":"2008.01077","repositories_listed":0,"syntology":null},{"url":null,"slug":"siamese-x-vector-reconstruction-for-domain","title":"Siamese x-vector reconstruction for domain adapted speaker recognition","date":"2020-07-28","arxiv_id":"2007.14146","repositories_listed":0,"syntology":null}],"record_sha256":"d9a8631e0fc2c3ac2c0fc058f485bc12c0066c085c0abd9f322ac3e23260531e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}