{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speaker-recognition/papers/2","list_of":"/task/speaker-recognition","task":"Speaker Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":5,"rows_per_page":100,"rows":[101,200],"of":435,"counts":{"archive_papers_tagged":435,"with_a_code_link":102,"where_syntology_ran_a_sample":17,"not_listed_spam_title":0,"listed":435,"listed_where_code_ran":17,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":15,"every_run_a_failure_of_syntologys_instrument":2,"listed_with_a_run_with_no_instrument_failure":15,"listed_every_run_a_failure_of_syntologys_instrument":2,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speaker-recognition","prev":"/task/speaker-recognition","next":"/task/speaker-recognition/papers/3","papers":[{"url":"/paper/frame-level-speaker-embeddings-for-text","slug":"frame-level-speaker-embeddings-for-text","title":"Frame-level speaker embeddings for text-independent speaker recognition and analysis of end-to-end model","date":"2018-09-12","arxiv_id":"1809.04437","repositories_listed":1,"syntology":null},{"url":"/paper/unified-hypersphere-embedding-for-speaker","slug":"unified-hypersphere-embedding-for-speaker","title":"Unified Hypersphere Embedding for Speaker Recognition","date":"2018-07-22","arxiv_id":"1807.08312","repositories_listed":1,"syntology":null},{"url":null,"slug":"an-exploration-of-ecapa-tdnn-and-x-vector","title":"An Exploration of ECAPA-TDNN and x-vector Speaker Representations in Zero-shot Multi-speaker TTS","date":"2025-06-25","arxiv_id":"2506.20190","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-evaluation-of-deep-learning-1","title":"A Comparative Evaluation of Deep Learning Models for Speech Enhancement in Real-World Noisy Environments","date":"2025-06-17","arxiv_id":"2506.15000","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-speaker-invariant-visual-features","title":"Learning Speaker-Invariant Visual Features for Lipreading","date":"2025-06-09","arxiv_id":"2506.07572","repositories_listed":0,"syntology":null},{"url":null,"slug":"rhythm-features-for-speaker-identification","title":"Rhythm Features for Speaker Identification","date":"2025-06-07","arxiv_id":"2506.06834","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-speech-source-tracing-using-metric","title":"Synthetic Speech Source Tracing using Metric Learning","date":"2025-06-03","arxiv_id":"2506.02590","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-reasonable-effectiveness-of","title":"Investigating the Reasonable Effectiveness of Speaker Pre-Trained Models and their Synergistic Power for SingMOS Prediction","date":"2025-06-02","arxiv_id":"2506.02232","repositories_listed":0,"syntology":null},{"url":null,"slug":"laspa-language-agnostic-speaker","title":"LASPA: Language Agnostic Speaker Disentanglement with Prefix-Tuned Cross-Attention","date":"2025-06-02","arxiv_id":"2506.02083","repositories_listed":0,"syntology":null},{"url":null,"slug":"source-tracing-of-synthetic-speech-systems","title":"Source Tracing of Synthetic Speech Systems Through Paralinguistic Pre-Trained Representations","date":"2025-06-01","arxiv_id":"2506.01157","repositories_listed":0,"syntology":null},{"url":null,"slug":"pretraining-multi-speaker-identification-for","title":"Pretraining Multi-Speaker Identification for Neural Speaker Diarization","date":"2025-05-30","arxiv_id":"2505.24545","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-abc-frontend-audio-systems-for","title":"Analysis of ABC Frontend Audio Systems for the NIST-SRE24","date":"2025-05-21","arxiv_id":"2505.15320","repositories_listed":0,"syntology":null},{"url":null,"slug":"socov-semi-orthogonal-parametric-pooling-of","title":"SoCov: Semi-Orthogonal Parametric Pooling of Covariance Matrix for Speaker Recognition","date":"2025-04-23","arxiv_id":"2504.16441","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-dialect-gaps-to-identity-maps-tackling","title":"From Dialect Gaps to Identity Maps: Tackling Variability in Speaker Verification","date":"2025-04-21","arxiv_id":"2505.04629","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-to-image-encoding-for-improved-voice","title":"Audio-to-Image Encoding for Improved Voice Characteristic Detection Using Deep Convolutional Neural Networks","date":"2025-03-07","arxiv_id":"2503.05929","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-modelling-for-speaker-diarization-in","title":"Language Modelling for Speaker Diarization in Telephonic Interviews","date":"2025-01-28","arxiv_id":"2501.17893","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxvietnam-a-large-scale-multi-genre-dataset","title":"VoxVietnam: a Large-Scale Multi-Genre Dataset for Vietnamese Speaker Recognition","date":"2024-12-31","arxiv_id":"2501.00328","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-prosodic-signatures-via-speech","title":"Investigating Prosodic Signatures via Speech Pre-Trained Models for Audio Deepfake Source Attribution","date":"2024-12-23","arxiv_id":"2412.17796","repositories_listed":0,"syntology":null},{"url":null,"slug":"study-on-inter-and-intra-speaker-variability","title":"Study on Inter and Intra Speaker Variability in Speaker Recognition","date":"2024-11-12","arxiv_id":"2411.07754","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-view-multi-task-modeling-with-speech","title":"Multi-View Multi-Task Modeling with Speech Foundation Models for Speech Forensic Tasks","date":"2024-10-16","arxiv_id":"2410.12947","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-speaker-representation-for","title":"Investigation of Speaker Representation for Target-Speaker Speech Processing","date":"2024-10-15","arxiv_id":"2410.11243","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-ocon-model-an-old-but-green-solution-for","title":"The OCON model: an old but green solution for distributable supervised classification for acoustic monitoring in smart cities","date":"2024-10-05","arxiv_id":"2410.04098","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-open-set-speaker-identification","title":"Enhancing Open-Set Speaker Identification through Rapid Tuning with Speaker Reciprocal Points and Negative Sample","date":"2024-09-24","arxiv_id":"2409.15742","repositories_listed":0,"syntology":null},{"url":null,"slug":"avengers-assemble-amalgamation-of-non","title":"Avengers Assemble: Amalgamation of Non-Semantic Features for Depression Detection","date":"2024-09-22","arxiv_id":"2409.14312","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-music-foundation-models-better-at-singing","title":"Are Music Foundation Models Better at Singing Voice Deepfake Detection? Far-Better Fuse them with Speech Foundation Models","date":"2024-09-21","arxiv_id":"2409.14131","repositories_listed":0,"syntology":null},{"url":null,"slug":"obovox-far-field-speaker-recognition-a-novel","title":"oboVox Far Field Speaker Recognition: A Novel Data Augmentation Approach with Pretrained Models","date":"2024-09-16","arxiv_id":"2409.10240","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-ipl-unsupervised-learning-of-speaker","title":"Speaker-IPL: Unsupervised Learning of Speaker Characteristics with i-Vector based Pseudo-Labels","date":"2024-09-16","arxiv_id":"2409.10791","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-speech-synthesis-in-the-wild","title":"Text-To-Speech Synthesis In The Wild","date":"2024-09-13","arxiv_id":"2409.08711","repositories_listed":0,"syntology":null},{"url":null,"slug":"recursive-attentive-pooling-for-extracting","title":"Recursive Attentive Pooling for Extracting Speaker Embeddings from Multi-Speaker Recordings","date":"2024-08-30","arxiv_id":"2408.17142","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-voxceleb-speaker-recognition-challenge-a","title":"The VoxCeleb Speaker Recognition Challenge: A Retrospective","date":"2024-08-27","arxiv_id":"2408.14886","repositories_listed":0,"syntology":null},{"url":null,"slug":"convexity-based-pruning-of-speech","title":"Convexity-based Pruning of Speech Representation Models","date":"2024-08-16","arxiv_id":"2408.11858","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-term-conversation-analysis-privacy","title":"Long-Term Conversation Analysis: Privacy-Utility Trade-off under Noise and Reverberation","date":"2024-08-01","arxiv_id":"2408.00382","repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-of-speaker-modeling-and-its","title":"Overview of Speaker Modeling and Its Applications: From the Lens of Deep Speaker Representation Learning","date":"2024-07-21","arxiv_id":"2407.15188","repositories_listed":0,"syntology":null},{"url":null,"slug":"team-hyu-asml-robovox-sp-cup-2024-system","title":"Team HYU ASML ROBOVOX SP Cup 2024 System Description","date":"2024-07-16","arxiv_id":"2407.11365","repositories_listed":0,"syntology":null},{"url":null,"slug":"phonetic-richness-for-improved-automatic","title":"Phonetic Richness for Improved Automatic Speaker Verification","date":"2024-07-10","arxiv_id":"2407.08017","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-speech-unit-selection-for-textless","title":"Analyzing Speech Unit Selection for Textless Speech-to-Speech Translation","date":"2024-07-08","arxiv_id":"2407.18332","repositories_listed":0,"syntology":null},{"url":null,"slug":"we-need-variations-in-speech-synthesis-sub","title":"We Need Variations in Speech Generation: Sub-center Modelling for Speaker Embeddings","date":"2024-07-05","arxiv_id":"2407.04291","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-source-conversational-ai-with","title":"Open-Source Conversational AI with SpeechBrain 1.0","date":"2024-06-29","arxiv_id":"2407.00463","repositories_listed":0,"syntology":null},{"url":null,"slug":"cec-a-noisy-label-detection-method-for","title":"CEC: A Noisy Label Detection Method for Speaker Recognition","date":"2024-06-19","arxiv_id":"2406.13268","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenging-margin-based-speaker-embedding","title":"Challenging margin-based speaker embedding extractors by using the variational information bottleneck","date":"2024-06-18","arxiv_id":"2406.12622","repositories_listed":0,"syntology":null},{"url":null,"slug":"persona-an-application-for-emotion","title":"PERSONA: An Application for Emotion Recognition, Gender Recognition and Age Estimation","date":"2024-06-10","arxiv_id":"2406.06781","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-reasonable-effectiveness-of-speaker","title":"The Reasonable Effectiveness of Speaker Embeddings for Violence Detection","date":"2024-06-10","arxiv_id":"2406.06798","repositories_listed":0,"syntology":null},{"url":null,"slug":"fill-in-the-gap-combining-self-supervised","title":"Fill in the Gap! Combining Self-supervised Representation Learning with Neural Audio Synthesis for Speech Inpainting","date":"2024-05-30","arxiv_id":"2405.20101","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-characterization-by-means-of","title":"Speaker Characterization by means of Attention Pooling","date":"2024-05-07","arxiv_id":"2405.04096","repositories_listed":0,"syntology":null},{"url":null,"slug":"who-is-authentic-speaker","title":"Who is Authentic Speaker","date":"2024-04-30","arxiv_id":"2405.00248","repositories_listed":0,"syntology":null},{"url":null,"slug":"artificial-neural-networks-to-recognize","title":"Artificial Neural Networks to Recognize Speakers Division from Continuous Bengali Speech","date":"2024-04-18","arxiv_id":"2404.15168","repositories_listed":0,"syntology":null},{"url":null,"slug":"timit-speaker-profiling-a-comparison-of-multi","title":"TIMIT Speaker Profiling: A Comparison of Multi-task learning and Single-task learning Approaches","date":"2024-04-18","arxiv_id":"2404.12077","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-conversion-augmentation-for-speaker","title":"Voice Conversion Augmentation for Speaker Recognition on Defective Datasets","date":"2024-04-01","arxiv_id":"2404.00863","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymmetric-and-trial-dependent-modeling-the","title":"Asymmetric and trial-dependent modeling: the contribution of LIA to SdSV Challenge Task 2","date":"2024-03-28","arxiv_id":"2403.19634","repositories_listed":0,"syntology":null},{"url":null,"slug":"cosine-scoring-with-uncertainty-for-neural","title":"Cosine Scoring with Uncertainty for Neural Speaker Embedding","date":"2024-03-11","arxiv_id":"2403.06404","repositories_listed":0,"syntology":null},{"url":null,"slug":"post-training-embedding-alignment-for","title":"Post-Training Embedding Alignment for Decoupling Enrollment and Runtime Speaker Recognition Models","date":"2024-01-23","arxiv_id":"2401.12440","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxceleb-esp-preliminary-experiments","title":"Voxceleb-ESP: preliminary experiments detecting Spanish celebrities from their voices","date":"2023-12-20","arxiv_id":"2401.09441","repositories_listed":0,"syntology":null},{"url":null,"slug":"vulnerability-of-automatic-identity","title":"Vulnerability of Automatic Identity Recognition to Audio-Visual Deepfakes","date":"2023-11-29","arxiv_id":"2311.17655","repositories_listed":0,"syntology":null},{"url":null,"slug":"phonetic-aware-speaker-embedding-for-far","title":"Phonetic-aware speaker embedding for far-field speaker verification","date":"2023-11-27","arxiv_id":"2311.15627","repositories_listed":0,"syntology":null},{"url":null,"slug":"parrot-trained-adversarial-examples-pushing","title":"Parrot-Trained Adversarial Examples: Pushing the Practicality of Black-Box Audio Attacks against Speaker Recognition Models","date":"2023-11-13","arxiv_id":"2311.07780","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-agreement-in-multi-party","title":"Detecting Agreement in Multi-party Conversational AI","date":"2023-11-06","arxiv_id":"2311.03026","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalizing-keyword-spotting-with-speaker","title":"Personalizing Keyword Spotting with Speaker Information","date":"2023-11-06","arxiv_id":"2311.03419","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-neural-networks-for-automatic-speaker","title":"Deep Neural Networks for Automatic Speaker Recognition Do Not Learn Supra-Segmental Temporal Features","date":"2023-11-01","arxiv_id":"2311.00489","repositories_listed":0,"syntology":null},{"url":null,"slug":"unix-encoder-a-universal-x-channel-speech","title":"UniX-Encoder: A Universal $X$-Channel Speech Encoder for Ad-Hoc Microphone Array Speech Processing","date":"2023-10-25","arxiv_id":"2310.16367","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-oriented-manipulation-of-speaker","title":"Privacy-oriented manipulation of speaker representations","date":"2023-10-10","arxiv_id":"2310.06652","repositories_listed":0,"syntology":null},{"url":null,"slug":"thech-report-genuinization-of-speech-waveform","title":"Thech. Report: Genuinization of Speech waveform PMF for speaker detection spoofing and countermeasures","date":"2023-10-09","arxiv_id":"2310.05534","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-voice-and-content-with-self","title":"Disentangling Voice and Content with Self-Supervision for Speaker Recognition","date":"2023-10-02","arxiv_id":"2310.01128","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-morphing-two-identities-in-one-voice","title":"Voice Morphing: Two Identities in One Voice","date":"2023-09-05","arxiv_id":"2309.02404","repositories_listed":0,"syntology":null},{"url":null,"slug":"unisound-system-for-voxceleb-speaker","title":"UNISOUND System for VoxCeleb Speaker Recognition Challenge 2023","date":"2023-08-24","arxiv_id":"2308.12526","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-neural-network-backend-for-speaker","title":"Graph Neural Network Backend for Speaker Recognition","date":"2023-08-17","arxiv_id":"2308.08767","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-dku-msxf-speaker-verification-system-for","title":"The DKU-MSXF Speaker Verification System for the VoxCeleb Speaker Recognition Challenge 2023","date":"2023-08-17","arxiv_id":"2308.08766","repositories_listed":0,"syntology":null},{"url":null,"slug":"chinatelecom-system-description-to-voxceleb","title":"ChinaTelecom System Description to VoxCeleb Speaker Recognition Challenge 2023","date":"2023-08-16","arxiv_id":"2308.08181","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-id-r-d-voxceleb-speaker-recognition","title":"The ID R&D VoxCeleb Speaker Recognition Challenge 2023 System Description","date":"2023-08-16","arxiv_id":"2308.08294","repositories_listed":0,"syntology":null},{"url":null,"slug":"gist-aiter-speaker-diarization-system-for","title":"GIST-AiTeR Speaker Diarization System for VoxCeleb Speaker Recognition Challenge (VoxSRC) 2023","date":"2023-08-15","arxiv_id":"2308.07788","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-dku-msxf-diarization-system-for-the","title":"The DKU-MSXF Diarization System for the VoxCeleb Speaker Recognition Challenge 2023","date":"2023-08-15","arxiv_id":"2308.07595","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxsnap-x-large-speaker-verification-dataset","title":"VoxBlink: A Large Scale Speaker Verification Dataset on Camera","date":"2023-08-14","arxiv_id":"2308.07056","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-speaker-anonymization-of-acoustic","title":"On-Device Speaker Anonymization of Acoustic Embeddings for ASR based onFlexible Location Gradient Reversal Layer","date":"2023-07-25","arxiv_id":"2307.13343","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-integration-of-speech","title":"Exploring the Integration of Speech Separation and Recognition with Self-Supervised Learning Representation","date":"2023-07-23","arxiv_id":"2307.12231","repositories_listed":0,"syntology":null},{"url":null,"slug":"facial-landmark-detection-evaluation-on-mobio","title":"Facial Landmark Detection Evaluation on MOBIO Database","date":"2023-07-06","arxiv_id":"2307.03329","repositories_listed":0,"syntology":null},{"url":null,"slug":"voxwatch-an-open-set-speaker-recognition","title":"VoxWatch: An open-set speaker recognition benchmark on VoxCeleb","date":"2023-06-30","arxiv_id":"2307.00169","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-contrastive-learning-through","title":"Understanding Contrastive Learning Through the Lens of Margins","date":"2023-06-20","arxiv_id":"2306.11526","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-specific-thresholding-for-robust","title":"Meta-Learning Framework for End-to-End Imposter Identification in Unseen Speaker Recognition","date":"2023-06-01","arxiv_id":"2306.00952","repositories_listed":0,"syntology":null},{"url":null,"slug":"stt4sg-350-a-speech-corpus-for-all-swiss","title":"STT4SG-350: A Speech Corpus for All Swiss German Dialect Regions","date":"2023-05-30","arxiv_id":"2305.18855","repositories_listed":0,"syntology":null},{"url":null,"slug":"transforming-the-embeddings-a-lightweight","title":"Transforming the Embeddings: A Lightweight Technique for Speech Emotion Recognition Tasks","date":"2023-05-29","arxiv_id":"2305.18640","repositories_listed":0,"syntology":null},{"url":null,"slug":"ordered-and-binary-speaker-embedding","title":"Ordered and Binary Speaker Embedding","date":"2023-05-25","arxiv_id":"2305.16043","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-domain-adaptation-framework-for","title":"Generalized domain adaptation framework for parametric back-end in speaker recognition","date":"2023-05-24","arxiv_id":"2305.15567","repositories_listed":0,"syntology":null},{"url":null,"slug":"qfa2sr-query-free-adversarial-transfer","title":"QFA2SR: Query-Free Adversarial Transfer Attacks to Speaker Recognition Systems","date":"2023-05-23","arxiv_id":"2305.14097","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-of-pre-trained-speech-and","title":"A Comparative Study of Pre-trained Speech and Audio Embeddings for Speech Emotion Recognition","date":"2023-04-22","arxiv_id":"2304.11472","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-graph-feature-fusion-technique-for","title":"The Graph feature fusion technique for speaker recognition based on wav2vec2.0 framework","date":"2023-03-19","arxiv_id":"2303.10556","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-bias-and-fairness-in-deep-speaker","title":"A Study on Bias and Fairness In Deep Speaker Recognition","date":"2023-03-14","arxiv_id":"2303.08026","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-film-conditioning-gans-with-self","title":"Self-FiLM: Conditioning GANs with self-supervised representations for bandwidth extension based speaker recognition","date":"2023-03-07","arxiv_id":"2303.03657","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-recognition-in-realistic-scenario","title":"Speaker Recognition in Realistic Scenario Using Multimodal Data","date":"2023-02-25","arxiv_id":"2302.13033","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-framework-for-online","title":"A Reinforcement Learning Framework for Online Speaker Diarization","date":"2023-02-21","arxiv_id":"2302.10924","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-spectrum-transformation-attacks","title":"Interpretable Spectrum Transformation Attacks to Speaker Recognition","date":"2023-02-21","arxiv_id":"2302.10686","repositories_listed":0,"syntology":null},{"url":null,"slug":"speaker-and-language-change-detection-using","title":"Speaker and Language Change Detection using Wav2vec2 and Whisper","date":"2023-02-18","arxiv_id":"2302.09381","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-representation-learning-by-distilling","title":"Audio Representation Learning by Distilling Video as Privileged Information","date":"2023-02-06","arxiv_id":"2302.02845","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-speaker-embeddings-with","title":"Leveraging Speaker Embeddings with Adversarial Multi-task Learning for Age Group Classification","date":"2023-01-22","arxiv_id":"2301.09058","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-purpose-audio-visual-corpus-for-multi","title":"A Multi-Purpose Audio-Visual Corpus for Multi-Modal Persian Speech Recognition: the Arman-AV Dataset","date":"2023-01-21","arxiv_id":"2301.10180","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-newsbridge-telecom-sudparis-voxceleb","title":"The Newsbridge -Telecom SudParis VoxCeleb Speaker Recognition Challenge 2022 System Description","date":"2023-01-17","arxiv_id":"2301.07491","repositories_listed":0,"syntology":null},{"url":null,"slug":"introducing-model-inversion-attacks-on","title":"Introducing Model Inversion Attacks on Automatic Speaker Recognition","date":"2023-01-09","arxiv_id":"2301.03206","repositories_listed":0,"syntology":null},{"url":null,"slug":"slue-phase-2-a-benchmark-suite-of-diverse","title":"SLUE Phase-2: A Benchmark Suite of Diverse Spoken Language Understanding Tasks","date":"2022-12-20","arxiv_id":"2212.10525","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-deep-speaker-embeddings-for-speaker","title":"Probing Deep Speaker Embeddings for Speaker-related Tasks","date":"2022-12-14","arxiv_id":"2212.07068","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-speech-feature-fusion-algorithm-for","title":"A Novel Speech Feature Fusion Algorithm for Text-Independent Speaker Recognition","date":"2022-12-01","arxiv_id":"2212.00329","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-speech-feature-fusion-method-with-cross","title":"A new Speech Feature Fusion method with cross gate parallel CNN for Speaker Recognition","date":"2022-11-24","arxiv_id":"2211.13377","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-source-domain-adaptation-for-text-1","title":"Multi-source Domain Adaptation for Text-independent Forensic Speaker Recognition","date":"2022-11-17","arxiv_id":"2211.09913","repositories_listed":0,"syntology":null}],"record_sha256":"1d97fc775e297b3794dc7a73cd9332f357df50f654564722119b4e69e10c3d2c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}