{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/2","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":65,"rows_per_page":100,"rows":[101,200],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition","next":"/task/speech-recognition/papers/3","papers":[{"url":"/paper/automatic-speech-recognition-benchmark-for","slug":"automatic-speech-recognition-benchmark-for","title":"Automatic Speech Recognition Benchmark for Air-Traffic Communications","date":"2020-06-18","arxiv_id":"2006.10304","repositories_listed":3,"syntology":null},{"url":"/paper/distilling-knowledge-from-ensembles-of","slug":"distilling-knowledge-from-ensembles-of","title":"Distilling Knowledge from Ensembles of Acoustic Models for Joint CTC-Attention End-to-End Speech Recognition","date":"2020-05-19","arxiv_id":"2005.09310","repositories_listed":3,"syntology":null},{"url":"/paper/speech-recognition-and-multi-speaker","slug":"speech-recognition-and-multi-speaker","title":"Speech Recognition and Multi-Speaker Diarization of Long Conversations","date":"2020-05-16","arxiv_id":"2005.08072","repositories_listed":3,"syntology":null},{"url":"/paper/macsen-a-voice-assistant-for-speakers-of-a","slug":"macsen-a-voice-assistant-for-speakers-of-a","title":"Macsen: A Voice Assistant for Speakers of a Lesser Resourced Language","date":"2020-05-01","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/unsupervised-pretraining-transfers-well","slug":"unsupervised-pretraining-transfers-well","title":"Unsupervised pretraining transfers well across languages","date":"2020-02-07","arxiv_id":"2002.02848","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-pretraining-transfers-well#ran","syntology_url":"https://syntology.ai/paper/2002.02848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.02848"}},"official":{"repos":["facebookresearch/CPC_audio"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/non-intrusive-load-monitoring-with-an","slug":"non-intrusive-load-monitoring-with-an","title":"Improving Non-Intrusive Load Disaggregation through an Attention-Based Deep Neural Network","date":"2019-11-15","arxiv_id":"1912.00759","repositories_listed":3,"syntology":null},{"url":"/paper/espnet-tts-unified-reproducible-and","slug":"espnet-tts-unified-reproducible-and","title":"ESPnet-TTS: Unified, Reproducible, and Integratable Open Source End-to-End Text-to-Speech Toolkit","date":"2019-10-24","arxiv_id":"1910.10909","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/espnet-tts-unified-reproducible-and#ran","syntology_url":"https://syntology.ai/paper/1910.10909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.10909"}},"official":{"repos":["r9y9/wavenet_vocoder","espnet/espnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/vq-wav2vec-self-supervised-learning-of-1","slug":"vq-wav2vec-self-supervised-learning-of-1","title":"vq-wav2vec: Self-Supervised Learning of Discrete Speech Representations","date":"2019-10-12","arxiv_id":"1910.05453","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vq-wav2vec-self-supervised-learning-of-1#ran","syntology_url":"https://syntology.ai/paper/1910.05453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.05453"}},"official":null}},{"url":"/paper/videobert-a-joint-model-for-video-and","slug":"videobert-a-joint-model-for-video-and","title":"VideoBERT: A Joint Model for Video and Language Representation Learning","date":"2019-04-03","arxiv_id":"1904.01766","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videobert-a-joint-model-for-video-and#ran","syntology_url":"https://syntology.ai/paper/1904.01766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.01766"}},"official":null}},{"url":"/paper/model-unit-exploration-for-sequence-to","slug":"model-unit-exploration-for-sequence-to","title":"On the Choice of Modeling Unit for Sequence-to-Sequence Speech Recognition","date":"2019-02-05","arxiv_id":"1902.01955","repositories_listed":3,"syntology":null},{"url":"/paper/pansori-asr-corpus-generation-from-open","slug":"pansori-asr-corpus-generation-from-open","title":"Pansori: ASR Corpus Generation from Open Online Video Contents","date":"2018-12-23","arxiv_id":"1812.09798","repositories_listed":3,"syntology":null},{"url":"/paper/improved-speech-enhancement-with-the-wave-u","slug":"improved-speech-enhancement-with-the-wave-u","title":"Improved Speech Enhancement with the Wave-U-Net","date":"2018-11-27","arxiv_id":"1811.11307","repositories_listed":3,"syntology":null},{"url":"/paper/learning-to-detect-dysarthria-from-raw-speech","slug":"learning-to-detect-dysarthria-from-raw-speech","title":"Learning to detect dysarthria from raw speech","date":"2018-11-27","arxiv_id":"1811.11101","repositories_listed":3,"syntology":null},{"url":"/paper/attention-based-audio-visual-fusion-for","slug":"attention-based-audio-visual-fusion-for","title":"Attention-based Audio-Visual Fusion for Robust Automatic Speech Recognition","date":"2018-09-05","arxiv_id":"1809.01728","repositories_listed":3,"syntology":null},{"url":"/paper/big-little-net-an-efficient-multi-scale","slug":"big-little-net-an-efficient-multi-scale","title":"Big-Little Net: An Efficient Multi-Scale Feature Representation for Visual and Speech Recognition","date":"2018-07-10","arxiv_id":"1807.03848","repositories_listed":3,"syntology":null},{"url":"/paper/quaternion-recurrent-neural-networks","slug":"quaternion-recurrent-neural-networks","title":"Quaternion Recurrent Neural Networks","date":"2018-06-12","arxiv_id":"1806.04418","repositories_listed":3,"syntology":null},{"url":"/paper/mixed-precision-training-for-nlp-and-speech","slug":"mixed-precision-training-for-nlp-and-speech","title":"Mixed-Precision Training for NLP and Speech Recognition with OpenSeq2Seq","date":"2018-05-25","arxiv_id":"1805.10387","repositories_listed":3,"syntology":null},{"url":"/paper/returnn-as-a-generic-flexible-neural-toolkit","slug":"returnn-as-a-generic-flexible-neural-toolkit","title":"RETURNN as a Generic Flexible Neural Toolkit with Application to Translation and Speech Recognition","date":"2018-05-14","arxiv_id":"1805.05225","repositories_listed":3,"syntology":null},{"url":"/paper/ted-lium-3-twice-as-much-data-and-corpus","slug":"ted-lium-3-twice-as-much-data-and-corpus","title":"TED-LIUM 3: twice as much data and corpus repartition for experiments on speaker adaptation","date":"2018-05-12","arxiv_id":"1805.04699","repositories_listed":3,"syntology":null},{"url":"/paper/spoken-squad-a-study-of-mitigating-the-impact","slug":"spoken-squad-a-study-of-mitigating-the-impact","title":"Spoken SQuAD: A Study of Mitigating the Impact of Speech Recognition Errors on Listening Comprehension","date":"2018-04-01","arxiv_id":"1804.00320","repositories_listed":3,"syntology":null},{"url":"/paper/a-deep-relevance-matching-model-for-ad-hoc","slug":"a-deep-relevance-matching-model-for-ad-hoc","title":"A Deep Relevance Matching Model for Ad-hoc Retrieval","date":"2017-11-23","arxiv_id":"1711.08611","repositories_listed":3,"syntology":null},{"url":"/paper/unsupervised-learning-of-disentangled-and","slug":"unsupervised-learning-of-disentangled-and","title":"Unsupervised Learning of Disentangled and Interpretable Representations from Sequential Data","date":"2017-09-22","arxiv_id":"1709.07902","repositories_listed":3,"syntology":null},{"url":"/paper/transfer-learning-for-speech-recognition-on-a","slug":"transfer-learning-for-speech-recognition-on-a","title":"Transfer Learning for Speech Recognition on a Budget","date":"2017-06-01","arxiv_id":"1706.00290","repositories_listed":3,"syntology":null},{"url":"/paper/residual-lstm-design-of-a-deep-recurrent","slug":"residual-lstm-design-of-a-deep-recurrent","title":"Residual LSTM: Design of a Deep Recurrent Architecture for Distant Speech Recognition","date":"2017-01-10","arxiv_id":"1701.03360","repositories_listed":3,"syntology":null},{"url":"/paper/deep-learning-using-linear-support-vector","slug":"deep-learning-using-linear-support-vector","title":"Deep Learning using Linear Support Vector Machines","date":"2013-06-02","arxiv_id":"1306.0239","repositories_listed":3,"syntology":null},{"url":"/paper/mambattention-mamba-with-multi-head-attention","slug":"mambattention-mamba-with-multi-head-attention","title":"MambAttention: Mamba with Multi-Head Attention for Generalizable Single-Channel Speech Enhancement","date":"2025-07-01","arxiv_id":"2507.00966","repositories_listed":2,"syntology":null},{"url":"/paper/cosyvoice-3-towards-in-the-wild-speech","slug":"cosyvoice-3-towards-in-the-wild-speech","title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","date":"2025-05-23","arxiv_id":"2505.17589","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cosyvoice-3-towards-in-the-wild-speech#ran","syntology_url":"https://syntology.ai/paper/2505.17589","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17589"}},"official":{"repos":["funaudiollm/cosyvoice"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/whisper-lm-improving-asr-models-with-language","slug":"whisper-lm-improving-asr-models-with-language","title":"Whisper-LM: Improving ASR Models with Language Models for Low-Resource Languages","date":"2025-03-30","arxiv_id":"2503.23542","repositories_listed":2,"syntology":null},{"url":"/paper/emg2qwerty-a-large-dataset-with-baselines-for","slug":"emg2qwerty-a-large-dataset-with-baselines-for","title":"emg2qwerty: A Large Dataset with Baselines for Touch Typing using Surface Electromyography","date":"2024-10-26","arxiv_id":"2410.20081","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emg2qwerty-a-large-dataset-with-baselines-for#ran","syntology_url":"https://syntology.ai/paper/2410.20081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20081"}},"official":{"repos":["facebookresearch/emg2qwerty"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ichigo-mixed-modal-early-fusion-realtime","slug":"ichigo-mixed-modal-early-fusion-realtime","title":"Ichigo: Mixed-Modal Early-Fusion Realtime Voice Assistant","date":"2024-10-20","arxiv_id":"2410.15316","repositories_listed":2,"syntology":null},{"url":"/paper/recent-advances-in-speech-language-models-a","slug":"recent-advances-in-speech-language-models-a","title":"Recent Advances in Speech Language Models: A Survey","date":"2024-10-01","arxiv_id":"2410.03751","repositories_listed":2,"syntology":null},{"url":"/paper/vhasr-a-multimodal-speech-recognition-system","slug":"vhasr-a-multimodal-speech-recognition-system","title":"VHASR: A Multimodal Speech Recognition System With Vision Hotwords","date":"2024-10-01","arxiv_id":"2410.00822","repositories_listed":2,"syntology":null},{"url":"/paper/whisperner-unified-open-named-entity-and","slug":"whisperner-unified-open-named-entity-and","title":"WhisperNER: Unified Open Named Entity and Speech Recognition","date":"2024-09-12","arxiv_id":"2409.08107","repositories_listed":2,"syntology":null},{"url":"/paper/2407-21054","slug":"2407-21054","title":"Sentiment Reasoning for Healthcare","date":"2024-07-24","arxiv_id":"2407.21054","repositories_listed":2,"syntology":null},{"url":"/paper/text-based-detection-of-on-hold-scripts-in","slug":"text-based-detection-of-on-hold-scripts-in","title":"Text-Based Detection of On-Hold Scripts in Contact Center Calls","date":"2024-07-13","arxiv_id":"2407.09849","repositories_listed":2,"syntology":null},{"url":"/paper/pretraining-end-to-end-keyword-search-with","slug":"pretraining-end-to-end-keyword-search-with","title":"Pretraining End-to-End Keyword Search with Automatically Discovered Acoustic Units","date":"2024-07-05","arxiv_id":"2407.04652","repositories_listed":2,"syntology":null},{"url":"/paper/gigaspeech-2-an-evolving-large-scale-and","slug":"gigaspeech-2-an-evolving-large-scale-and","title":"GigaSpeech 2: An Evolving, Large-Scale and Multi-domain ASR Corpus for Low-Resource Languages with Automated Crawling, Transcription and Refinement","date":"2024-06-17","arxiv_id":"2406.11546","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gigaspeech-2-an-evolving-large-scale-and#ran","syntology_url":"https://syntology.ai/paper/2406.11546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11546"}},"official":{"repos":["SpeechColab/GigaSpeech2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diarizationlm-speaker-diarization-post","slug":"diarizationlm-speaker-diarization-post","title":"DiarizationLM: Speaker Diarization Post-Processing with Large Language Models","date":"2024-01-07","arxiv_id":"2401.03506","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diarizationlm-speaker-diarization-post#ran","syntology_url":"https://syntology.ai/paper/2401.03506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03506"}},"official":{"repos":["google/speaker-id"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-npu-aslp-liauto-system-description-for","slug":"the-npu-aslp-liauto-system-description-for","title":"The NPU-ASLP-LiAuto System Description for Visual Speech Recognition in CNVSRC 2023","date":"2024-01-07","arxiv_id":"2401.06788","repositories_listed":2,"syntology":null},{"url":"/paper/qwen-audio-advancing-universal-audio","slug":"qwen-audio-advancing-universal-audio","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","date":"2023-11-14","arxiv_id":"2311.07919","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/qwen-audio-advancing-universal-audio#ran","syntology_url":"https://syntology.ai/paper/2311.07919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07919"}},"official":{"repos":["qwenlm/qwen-audio"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-disfluency-detection-from","slug":"automatic-disfluency-detection-from","title":"Automatic Disfluency Detection from Untranscribed Speech","date":"2023-11-01","arxiv_id":"2311.00867","repositories_listed":2,"syntology":null},{"url":"/paper/distil-whisper-robust-knowledge-distillation","slug":"distil-whisper-robust-knowledge-distillation","title":"Distil-Whisper: Robust Knowledge Distillation via Large-Scale Pseudo Labelling","date":"2023-11-01","arxiv_id":"2311.00430","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/distil-whisper-robust-knowledge-distillation#ran","syntology_url":"https://syntology.ai/paper/2311.00430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.00430"}},"official":{"repos":["huggingface/distil-whisper"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/back-transcription-as-a-method-for-evaluating","slug":"back-transcription-as-a-method-for-evaluating","title":"Back Transcription as a Method for Evaluating Robustness of Natural Language Understanding Models to Speech Recognition Errors","date":"2023-10-25","arxiv_id":"2310.16609","repositories_listed":2,"syntology":null},{"url":"/paper/lauragpt-listen-attend-understand-and","slug":"lauragpt-listen-attend-understand-and","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","date":"2023-10-07","arxiv_id":"2310.04673","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lauragpt-listen-attend-understand-and#ran","syntology_url":"https://syntology.ai/paper/2310.04673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04673"}},"official":null}},{"url":"/paper/librispeech-pc-benchmark-for-evaluation-of","slug":"librispeech-pc-benchmark-for-evaluation-of","title":"LibriSpeech-PC: Benchmark for Evaluation of Punctuation and Capitalization Capabilities of end-to-end ASR Models","date":"2023-10-04","arxiv_id":"2310.02943","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/librispeech-pc-benchmark-for-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2310.02943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02943"}},"official":null}},{"url":"/paper/encodecmae-leveraging-neural-codecs-for","slug":"encodecmae-leveraging-neural-codecs-for","title":"EnCodecMAE: Leveraging neural codecs for universal audio representation learning","date":"2023-09-14","arxiv_id":"2309.07391","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/encodecmae-leveraging-neural-codecs-for#ran","syntology_url":"https://syntology.ai/paper/2309.07391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07391"}},"official":{"repos":["habla-liaa/encodecmae"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/promptasr-for-contextualized-asr-with","slug":"promptasr-for-contextualized-asr-with","title":"PromptASR for contextualized ASR with controllable style","date":"2023-09-14","arxiv_id":"2309.07414","repositories_listed":2,"syntology":null},{"url":"/paper/conformer-based-target-speaker-automatic","slug":"conformer-based-target-speaker-automatic","title":"Conformer-based Target-Speaker Automatic Speech Recognition for Single-Channel Audio","date":"2023-08-09","arxiv_id":"2308.05218","repositories_listed":2,"syntology":null},{"url":"/paper/learning-multi-modal-representations-by","slug":"learning-multi-modal-representations-by","title":"Learning Multi-modal Representations by Watching Hundreds of Surgical Video Lectures","date":"2023-07-27","arxiv_id":"2307.15220","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/learning-multi-modal-representations-by#ran","syntology_url":"https://syntology.ai/paper/2307.15220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.15220"}},"official":{"repos":["camma-public/peskavlp","camma-public/surgvlp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/ivrit-ai-a-comprehensive-dataset-of-hebrew","slug":"ivrit-ai-a-comprehensive-dataset-of-hebrew","title":"ivrit.ai: A Comprehensive Dataset of Hebrew Speech for AI Research and Development","date":"2023-07-17","arxiv_id":"2307.08720","repositories_listed":2,"syntology":null},{"url":"/paper/quilt-1m-one-million-image-text-pairs-for-1","slug":"quilt-1m-one-million-image-text-pairs-for-1","title":"Quilt-1M: One Million Image-Text Pairs for Histopathology","date":"2023-06-20","arxiv_id":"2306.11207","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quilt-1m-one-million-image-text-pairs-for-1#ran","syntology_url":"https://syntology.ai/paper/2306.11207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11207"}},"official":{"repos":["wisdomikezogwo/quilt1m"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unit-based-speech-to-speech-translation","slug":"unit-based-speech-to-speech-translation","title":"Textless Speech-to-Speech Translation With Limited Parallel Data","date":"2023-05-24","arxiv_id":"2305.15405","repositories_listed":2,"syntology":null},{"url":"/paper/exploring-energy-based-language-models-with","slug":"exploring-energy-based-language-models-with","title":"Exploring Energy-based Language Models with Different Architectures and Training Methods for Speech Recognition","date":"2023-05-22","arxiv_id":"2305.12676","repositories_listed":2,"syntology":null},{"url":"/paper/a-new-benchmark-of-aphasia-speech-recognition","slug":"a-new-benchmark-of-aphasia-speech-recognition","title":"A New Benchmark of Aphasia Speech Recognition and Detection Based on E-Branchformer and Multi-task Learning","date":"2023-05-19","arxiv_id":"2305.13331","repositories_listed":2,"syntology":null},{"url":"/paper/a-comparative-study-on-e-branchformer-vs","slug":"a-comparative-study-on-e-branchformer-vs","title":"A Comparative Study on E-Branchformer vs Conformer in Speech Recognition, Translation, and Understanding Tasks","date":"2023-05-18","arxiv_id":"2305.11073","repositories_listed":2,"syntology":null},{"url":"/paper/x-llm-bootstrapping-advanced-large-language","slug":"x-llm-bootstrapping-advanced-large-language","title":"X-LLM: Bootstrapping Advanced Large Language Models by Treating Multi-Modalities as Foreign Languages","date":"2023-05-07","arxiv_id":"2305.04160","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/x-llm-bootstrapping-advanced-large-language#ran","syntology_url":"https://syntology.ai/paper/2305.04160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04160"}},"official":null}},{"url":"/paper/reproducibility-is-nothing-without","slug":"reproducibility-is-nothing-without","title":"When Good and Reproducible Results are a Giant with Feet of Clay: The Importance of Software Quality in NLP","date":"2023-03-28","arxiv_id":"2303.16166","repositories_listed":2,"syntology":null},{"url":"/paper/auto-avsr-audio-visual-speech-recognition","slug":"auto-avsr-audio-visual-speech-recognition","title":"Auto-AVSR: Audio-Visual Speech Recognition with Automatic Labels","date":"2023-03-25","arxiv_id":"2303.14307","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/auto-avsr-audio-visual-speech-recognition#ran","syntology_url":"https://syntology.ai/paper/2303.14307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14307"}},"official":{"repos":["mpc001/auto_avsr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-transfer-from-pre-trained-language","slug":"knowledge-transfer-from-pre-trained-language","title":"Knowledge Transfer from Pre-trained Language Models to Cif-based Speech Recognizers via Hierarchical Distillation","date":"2023-01-30","arxiv_id":"2301.13003","repositories_listed":2,"syntology":null},{"url":"/paper/unsupervised-data-selection-for-tts-using","slug":"unsupervised-data-selection-for-tts-using","title":"Unsupervised Data Selection for TTS: Using Arabic Broadcast News as a Case Study","date":"2023-01-22","arxiv_id":"2301.09099","repositories_listed":2,"syntology":null},{"url":"/paper/gpu-accelerated-guided-source-separation-for","slug":"gpu-accelerated-guided-source-separation-for","title":"GPU-accelerated Guided Source Separation for Meeting Transcription","date":"2022-12-10","arxiv_id":"2212.05271","repositories_listed":2,"syntology":null},{"url":"/paper/a-persian-asr-based-ser-modification-of","slug":"a-persian-asr-based-ser-modification-of","title":"A Persian ASR-based SER: Modification of Sharif Emotional Speech Database and Investigation of Persian Text Corpora","date":"2022-11-18","arxiv_id":"2211.09956","repositories_listed":2,"syntology":null},{"url":"/paper/esb-a-benchmark-for-multi-domain-end-to-end","slug":"esb-a-benchmark-for-multi-domain-end-to-end","title":"ESB: A Benchmark For Multi-Domain End-to-End Speech Recognition","date":"2022-10-24","arxiv_id":"2210.13352","repositories_listed":2,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/esb-a-benchmark-for-multi-domain-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2210.13352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13352"}},"official":null}},{"url":"/paper/speechut-bridging-speech-and-text-with-hidden","slug":"speechut-bridging-speech-and-text-with-hidden","title":"SpeechUT: Bridging Speech and Text with Hidden-Unit for Encoder-Decoder Based Speech-Text Pre-training","date":"2022-10-07","arxiv_id":"2210.03730","repositories_listed":2,"syntology":null},{"url":"/paper/cmgan-conformer-based-metric-gan-for-monaural","slug":"cmgan-conformer-based-metric-gan-for-monaural","title":"CMGAN: Conformer-Based Metric-GAN for Monaural Speech Enhancement","date":"2022-09-22","arxiv_id":"2209.11112","repositories_listed":2,"syntology":null},{"url":"/paper/improved-open-source-automatic-subtitling-for","slug":"improved-open-source-automatic-subtitling-for","title":"Improved Open Source Automatic Subtitling for Lecture Videos","date":"2022-09-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/improving-mandarin-speech-recogntion-with","slug":"improving-mandarin-speech-recogntion-with","title":"Improving Mandarin Speech Recogntion with Block-augmented Transformer","date":"2022-07-24","arxiv_id":"2207.11697","repositories_listed":2,"syntology":null},{"url":"/paper/tevr-improving-speech-recognition-by-token","slug":"tevr-improving-speech-recognition-by-token","title":"TEVR: Improving Speech Recognition by Token Entropy Variance Reduction","date":"2022-06-25","arxiv_id":"2206.12693","repositories_listed":2,"syntology":null},{"url":"/paper/paraformer-fast-and-accurate-parallel","slug":"paraformer-fast-and-accurate-parallel","title":"Paraformer: Fast and Accurate Parallel Transformer for Non-autoregressive End-to-End Speech Recognition","date":"2022-06-16","arxiv_id":"2206.08317","repositories_listed":2,"syntology":null},{"url":"/paper/soundspaces-2-0-a-simulation-platform-for","slug":"soundspaces-2-0-a-simulation-platform-for","title":"SoundSpaces 2.0: A Simulation Platform for Visual-Acoustic Learning","date":"2022-06-16","arxiv_id":"2206.08312","repositories_listed":2,"syntology":null},{"url":"/paper/towards-understanding-and-mitigating-audio","slug":"towards-understanding-and-mitigating-audio","title":"Towards Understanding and Mitigating Audio Adversarial Examples for Speaker Recognition","date":"2022-06-07","arxiv_id":"2206.03393","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-understanding-and-mitigating-audio#ran","syntology_url":"https://syntology.ai/paper/2206.03393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.03393"}},"official":null}},{"url":"/paper/paddlespeech-an-easy-to-use-all-in-one-speech-1","slug":"paddlespeech-an-easy-to-use-all-in-one-speech-1","title":"PaddleSpeech: An Easy-to-Use All-in-One Speech Toolkit","date":"2022-05-20","arxiv_id":"2205.12007","repositories_listed":2,"syntology":null},{"url":"/paper/how-does-pre-trained-wav2vec2-0-perform-on","slug":"how-does-pre-trained-wav2vec2-0-perform-on","title":"How Does Pre-trained Wav2Vec 2.0 Perform on Domain Shifted ASR? An Extensive Benchmark on Air Traffic Control Communications","date":"2022-03-31","arxiv_id":"2203.16822","repositories_listed":2,"syntology":null},{"url":"/paper/recent-improvements-of-asr-models-in-the-face","slug":"recent-improvements-of-asr-models-in-the-face","title":"Recent improvements of ASR models in the face of adversarial attacks","date":"2022-03-29","arxiv_id":"2203.16536","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recent-improvements-of-asr-models-in-the-face#ran","syntology_url":"https://syntology.ai/paper/2203.16536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16536"}},"official":{"repos":["raphaelolivier/robust_speech"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/listen-adapt-better-wer-source-free-single","slug":"listen-adapt-better-wer-source-free-single","title":"Listen, Adapt, Better WER: Source-free Single-utterance Test-time Adaptation for Automatic Speech Recognition","date":"2022-03-27","arxiv_id":"2203.14222","repositories_listed":2,"syntology":null},{"url":"/paper/a-3-t-alignment-aware-acoustic-and-text","slug":"a-3-t-alignment-aware-acoustic-and-text","title":"A$^3$T: Alignment-Aware Acoustic and Text Pretraining for Speech Synthesis and Editing","date":"2022-03-18","arxiv_id":"2203.09690","repositories_listed":2,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/a-3-t-alignment-aware-acoustic-and-text#ran","syntology_url":"https://syntology.ai/paper/2203.09690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09690"}},"official":null}},{"url":"/paper/visual-speech-recognition-for-multiple","slug":"visual-speech-recognition-for-multiple","title":"Visual Speech Recognition for Multiple Languages in the Wild","date":"2022-02-26","arxiv_id":"2202.13084","repositories_listed":2,"syntology":null},{"url":"/paper/learning-audio-visual-speech-representation-1","slug":"learning-audio-visual-speech-representation-1","title":"Learning Audio-Visual Speech Representation by Masked Multimodal Cluster Prediction","date":"2022-01-05","arxiv_id":"2201.02184","repositories_listed":2,"syntology":null},{"url":"/paper/xls-r-self-supervised-cross-lingual-speech","slug":"xls-r-self-supervised-cross-lingual-speech","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","date":"2021-11-17","arxiv_id":"2111.09296","repositories_listed":2,"syntology":null},{"url":"/paper/attention-based-multi-hypothesis-fusion-for","slug":"attention-based-multi-hypothesis-fusion-for","title":"Attention-based Multi-hypothesis Fusion for Speech Summarization","date":"2021-11-16","arxiv_id":"2111.08201","repositories_listed":2,"syntology":null},{"url":"/paper/coraa-a-large-corpus-of-spontaneous-and","slug":"coraa-a-large-corpus-of-spontaneous-and","title":"CORAA: a large corpus of spontaneous and prepared speech manually validated for speech recognition in Brazilian Portuguese","date":"2021-10-14","arxiv_id":"2110.15731","repositories_listed":2,"syntology":null},{"url":"/paper/bertraffic-a-robust-bert-based-approach-for","slug":"bertraffic-a-robust-bert-based-approach-for","title":"BERTraffic: BERT-based Joint Speaker Role and Speaker Change Detection for Air Traffic Control Communications","date":"2021-10-12","arxiv_id":"2110.05781","repositories_listed":2,"syntology":null},{"url":"/paper/interactive-feature-fusion-for-end-to-end","slug":"interactive-feature-fusion-for-end-to-end","title":"Interactive Feature Fusion for End-to-End Noise-Robust Speech Recognition","date":"2021-10-11","arxiv_id":"2110.05267","repositories_listed":2,"syntology":null},{"url":"/paper/k-wav2vec-2-0-automatic-speech-recognition","slug":"k-wav2vec-2-0-automatic-speech-recognition","title":"K-Wav2vec 2.0: Automatic Speech Recognition based on Joint Decoding of Graphemes and Syllables","date":"2021-10-11","arxiv_id":"2110.05172","repositories_listed":2,"syntology":null},{"url":"/paper/wenetspeech-a-10000-hours-multi-domain","slug":"wenetspeech-a-10000-hours-multi-domain","title":"WenetSpeech: A 10000+ Hours Multi-domain Mandarin Corpus for Speech Recognition","date":"2021-10-07","arxiv_id":"2110.03370","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wenetspeech-a-10000-hours-multi-domain#ran","syntology_url":"https://syntology.ai/paper/2110.03370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.03370"}},"official":{"repos":["wenet-e2e/wenetspeech"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/simple-and-effective-zero-shot-cross-lingual","slug":"simple-and-effective-zero-shot-cross-lingual","title":"Simple and Effective Zero-shot Cross-lingual Phoneme Recognition","date":"2021-09-23","arxiv_id":"2109.11680","repositories_listed":2,"syntology":null},{"url":"/paper/sveva-fair-a-framework-for-evaluating","slug":"sveva-fair-a-framework-for-evaluating","title":"SVEva Fair: A Framework for Evaluating Fairness in Speaker Verification","date":"2021-07-26","arxiv_id":"2107.12049","repositories_listed":2,"syntology":null},{"url":"/paper/a-comparison-of-methods-for-oov-word","slug":"a-comparison-of-methods-for-oov-word","title":"A Comparison of Methods for OOV-word Recognition on a New Public Dataset","date":"2021-07-16","arxiv_id":"2107.08091","repositories_listed":2,"syntology":null},{"url":"/paper/clsril-23-cross-lingual-speech","slug":"clsril-23-cross-lingual-speech","title":"CLSRIL-23: Cross Lingual Speech Representations for Indic Languages","date":"2021-07-15","arxiv_id":"2107.07402","repositories_listed":2,"syntology":null},{"url":"/paper/meshrir-a-dataset-of-room-impulse-responses","slug":"meshrir-a-dataset-of-room-impulse-responses","title":"MeshRIR: A Dataset of Room Impulse Responses on Meshed Grid Points For Evaluating Sound Field Analysis and Synthesis Methods","date":"2021-06-21","arxiv_id":"2106.10801","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meshrir-a-dataset-of-room-impulse-responses#ran","syntology_url":"https://syntology.ai/paper/2106.10801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.10801"}},"official":{"repos":["sh01k/MeshRIR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lightweight-adapter-tuning-for-multilingual","slug":"lightweight-adapter-tuning-for-multilingual","title":"Lightweight Adapter Tuning for Multilingual Speech Translation","date":"2021-06-02","arxiv_id":"2106.01463","repositories_listed":2,"syntology":null},{"url":"/paper/fedscale-benchmarking-model-and-system","slug":"fedscale-benchmarking-model-and-system","title":"FedScale: Benchmarking Model and System Performance of Federated Learning at Scale","date":"2021-05-24","arxiv_id":"2105.11367","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fedscale-benchmarking-model-and-system#ran","syntology_url":"https://syntology.ai/paper/2105.11367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.11367"}},"official":{"repos":["SymbioticLab/FedScale"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/exploiting-adapters-for-cross-lingual-low","slug":"exploiting-adapters-for-cross-lingual-low","title":"Exploiting Adapters for Cross-lingual Low-resource Speech Recognition","date":"2021-05-18","arxiv_id":"2105.11905","repositories_listed":2,"syntology":null},{"url":"/paper/emotion-recognition-from-speech-using-wav2vec","slug":"emotion-recognition-from-speech-using-wav2vec","title":"Emotion Recognition from Speech Using Wav2vec 2.0 Embeddings","date":"2021-04-08","arxiv_id":"2104.03502","repositories_listed":2,"syntology":null},{"url":"/paper/librispeech-transducer-model-with-internal","slug":"librispeech-transducer-model-with-internal","title":"Librispeech Transducer Model with Internal Language Model Prior Correction","date":"2021-04-07","arxiv_id":"2104.03006","repositories_listed":2,"syntology":null},{"url":"/paper/fdlp-spectrogram-capturing-speech-dynamics-in","slug":"fdlp-spectrogram-capturing-speech-dynamics-in","title":"Radically Old Way of Computing Spectra: Applications in End-to-End ASR","date":"2021-03-25","arxiv_id":"2103.14129","repositories_listed":2,"syntology":null},{"url":"/paper/domain-generalization-a-survey","slug":"domain-generalization-a-survey","title":"Domain Generalization: A Survey","date":"2021-03-03","arxiv_id":"2103.02503","repositories_listed":2,"syntology":null},{"url":"/paper/bembaspeech-a-speech-recognition-corpus-for","slug":"bembaspeech-a-speech-recognition-corpus-for","title":"BembaSpeech: A Speech Recognition Corpus for the Bemba Language","date":"2021-02-09","arxiv_id":"2102.04889","repositories_listed":2,"syntology":null},{"url":"/paper/understanding-the-tradeoffs-in-client-side","slug":"understanding-the-tradeoffs-in-client-side","title":"Understanding the Tradeoffs in Client-side Privacy for Downstream Speech Tasks","date":"2021-01-22","arxiv_id":"2101.08919","repositories_listed":2,"syntology":null},{"url":"/paper/kaleidoscope-an-efficient-learnable-1","slug":"kaleidoscope-an-efficient-learnable-1","title":"Kaleidoscope: An Efficient, Learnable Representation For All Structured Linear Maps","date":"2020-12-29","arxiv_id":"2012.14966","repositories_listed":2,"syntology":{"n":23,"n_ran":16,"n_constructed":9,"n_ran_checked":9,"n_instrument":7,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"16 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 7 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/kaleidoscope-an-efficient-learnable-1#ran","syntology_url":"https://syntology.ai/paper/2012.14966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.14966"}},"official":{"repos":["HazyResearch/butterfly","HazyResearch/learning-circuits"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":9,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}}],"record_sha256":"7e7936fb329f3e724e92c32540d7a0492cbdecbcf7834d17dee89350bfb168e3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}