{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/3","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":65,"rows_per_page":100,"rows":[201,300],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/2","next":"/task/speech-recognition/papers/4","papers":[{"url":"/paper/a-comprehensive-evaluation-of-incremental","slug":"a-comprehensive-evaluation-of-incremental","title":"A Comprehensive Evaluation of Incremental Speech Recognition and Diarization for Conversational AI","date":"2020-12-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/decentralizing-feature-extraction-with","slug":"decentralizing-feature-extraction-with","title":"Decentralizing Feature Extraction with Quantum Convolutional Neural Network for Automatic Speech Recognition","date":"2020-10-26","arxiv_id":"2010.13309","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/decentralizing-feature-extraction-with#ran","syntology_url":"https://syntology.ai/paper/2010.13309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.13309"}},"official":{"repos":["huckiyang/speech_quantum_dl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/towards-resistant-audio-adversarial-examples","slug":"towards-resistant-audio-adversarial-examples","title":"Towards Resistant Audio Adversarial Examples","date":"2020-10-14","arxiv_id":"2010.07190","repositories_listed":2,"syntology":null},{"url":"/paper/representation-learning-for-sequence-data-1","slug":"representation-learning-for-sequence-data-1","title":"Representation Learning for Sequence Data with Deep Autoencoding Predictive Components","date":"2020-10-07","arxiv_id":"2010.03135","repositories_listed":2,"syntology":{"n":29,"n_ran":22,"n_constructed":0,"n_ran_checked":22,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":22,"n_pointer_only":2,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 22 with no instrument failure: 0 honoured, 0 violated, 22 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/representation-learning-for-sequence-data-1#ran","syntology_url":"https://syntology.ai/paper/2010.03135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.03135"}},"official":{"repos":["JunwenBai/DAPC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed"]}}},{"url":"/paper/end-to-end-speech-recognition-and-disfluency","slug":"end-to-end-speech-recognition-and-disfluency","title":"End-to-End Speech Recognition and Disfluency Removal","date":"2020-09-22","arxiv_id":"2009.10298","repositories_listed":2,"syntology":null},{"url":"/paper/voicefilter-lite-streaming-targeted-voice","slug":"voicefilter-lite-streaming-targeted-voice","title":"VoiceFilter-Lite: Streaming Targeted Voice Separation for On-Device Speech Recognition","date":"2020-09-09","arxiv_id":"2009.04323","repositories_listed":2,"syntology":null},{"url":"/paper/computer-generated-music-for-tabletop-role","slug":"computer-generated-music-for-tabletop-role","title":"Computer-Generated Music for Tabletop Role-Playing Games","date":"2020-08-16","arxiv_id":"2008.07009","repositories_listed":2,"syntology":null},{"url":"/paper/pretraining-techniques-for-sequence-to","slug":"pretraining-techniques-for-sequence-to","title":"Pretraining Techniques for Sequence-to-Sequence Voice Conversion","date":"2020-08-07","arxiv_id":"2008.03088","repositories_listed":2,"syntology":null},{"url":"/paper/covost-2-a-massively-multilingual-speech-to","slug":"covost-2-a-massively-multilingual-speech-to","title":"CoVoST 2 and Massively Multilingual Speech-to-Text Translation","date":"2020-07-20","arxiv_id":"2007.10310","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/covost-2-a-massively-multilingual-speech-to#ran","syntology_url":"https://syntology.ai/paper/2007.10310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.10310"}},"official":{"repos":["facebookresearch/covost"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-lyrics-transcription-using-dilated","slug":"automatic-lyrics-transcription-using-dilated","title":"Automatic Lyrics Transcription using Dilated Convolutional Neural Networks with Self-Attention","date":"2020-07-13","arxiv_id":"2007.06486","repositories_listed":2,"syntology":null},{"url":"/paper/emotion-recognition-in-audio-and-video-using","slug":"emotion-recognition-in-audio-and-video-using","title":"Emotion Recognition in Audio and Video Using Deep Neural Networks","date":"2020-06-15","arxiv_id":"2006.08129","repositories_listed":2,"syntology":null},{"url":"/paper/an-adversarial-approach-for-explaining-the","slug":"an-adversarial-approach-for-explaining-the","title":"An Adversarial Approach for Explaining the Predictions of Deep Neural Networks","date":"2020-05-20","arxiv_id":"2005.10284","repositories_listed":2,"syntology":null},{"url":"/paper/discriminative-multi-modality-speech","slug":"discriminative-multi-modality-speech","title":"Discriminative Multi-modality Speech Recognition","date":"2020-05-12","arxiv_id":"2005.05592","repositories_listed":2,"syntology":null},{"url":"/paper/multilingual-twitter-corpus-and-baselines-for","slug":"multilingual-twitter-corpus-and-baselines-for","title":"Multilingual Twitter Corpus and Baselines for Evaluating Demographic Bias in Hate Speech Recognition","date":"2020-02-24","arxiv_id":"2002.10361","repositories_listed":2,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/multilingual-twitter-corpus-and-baselines-for#ran","syntology_url":"https://syntology.ai/paper/2002.10361","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.10361"}},"official":{"repos":["xiaoleihuang/Multilingual_Fairness_LREC"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/libri-light-a-benchmark-for-asr-with-limited","slug":"libri-light-a-benchmark-for-asr-with-limited","title":"Libri-Light: A Benchmark for ASR with Limited or No Supervision","date":"2019-12-17","arxiv_id":"1912.07875","repositories_listed":2,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/libri-light-a-benchmark-for-asr-with-limited#ran","syntology_url":"https://syntology.ai/paper/1912.07875","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.07875"}},"official":{"repos":["facebookresearch/libri-light"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/common-voice-a-massively-multilingual-speech","slug":"common-voice-a-massively-multilingual-speech","title":"Common Voice: A Massively-Multilingual Speech Corpus","date":"2019-12-13","arxiv_id":"1912.06670","repositories_listed":2,"syntology":null},{"url":"/paper/cat-crf-based-asr-toolkit","slug":"cat-crf-based-asr-toolkit","title":"CAT: CRF-based ASR Toolkit","date":"2019-11-20","arxiv_id":"1911.08747","repositories_listed":2,"syntology":null},{"url":"/paper/effectiveness-of-self-supervised-pre-training","slug":"effectiveness-of-self-supervised-pre-training","title":"Effectiveness of self-supervised pre-training for speech recognition","date":"2019-11-10","arxiv_id":"1911.03912","repositories_listed":2,"syntology":null},{"url":"/paper/confidence-estimation-for-black-box-automatic","slug":"confidence-estimation-for-black-box-automatic","title":"Confidence Estimation for Black Box Automatic Speech Recognition Systems Using Lattice Recurrent Neural Networks","date":"2019-10-25","arxiv_id":"1910.11933","repositories_listed":2,"syntology":null},{"url":"/paper/generative-pre-training-for-speech-with","slug":"generative-pre-training-for-speech-with","title":"Generative Pre-Training for Speech with Autoregressive Predictive Coding","date":"2019-10-23","arxiv_id":"1910.12607","repositories_listed":2,"syntology":null},{"url":"/paper/speech-vgg-a-deep-feature-extractor-for","slug":"speech-vgg-a-deep-feature-extractor-for","title":"Word-level Embeddings for Cross-Task Transfer Learning in Speech Processing","date":"2019-10-22","arxiv_id":"1910.09909","repositories_listed":2,"syntology":null},{"url":"/paper/distributed-learning-of-deep-neural-networks","slug":"distributed-learning-of-deep-neural-networks","title":"Distributed Learning of Deep Neural Networks using Independent Subnet Training","date":"2019-10-04","arxiv_id":"1910.02120","repositories_listed":2,"syntology":null},{"url":"/paper/dover-a-method-for-combining-diarization","slug":"dover-a-method-for-combining-diarization","title":"DOVER: A Method for Combining Diarization Outputs","date":"2019-09-17","arxiv_id":"1909.08090","repositories_listed":2,"syntology":null},{"url":"/paper/a-comparative-study-on-transformer-vs-rnn-in","slug":"a-comparative-study-on-transformer-vs-rnn-in","title":"A Comparative Study on Transformer vs RNN in Speech Applications","date":"2019-09-13","arxiv_id":"1909.06317","repositories_listed":2,"syntology":null},{"url":"/paper/ebpc-extended-bit-plane-compression-for-deep","slug":"ebpc-extended-bit-plane-compression-for-deep","title":"EBPC: Extended Bit-Plane Compression for Deep Neural Network Inference and Training Accelerators","date":"2019-08-30","arxiv_id":"1908.11645","repositories_listed":2,"syntology":null},{"url":"/paper/personal-vad-speaker-conditioned-voice","slug":"personal-vad-speaker-conditioned-voice","title":"Personal VAD: Speaker-Conditioned Voice Activity Detection","date":"2019-08-12","arxiv_id":"1908.04284","repositories_listed":2,"syntology":null},{"url":"/paper/delta-a-deep-learning-based-language","slug":"delta-a-deep-learning-based-language","title":"DELTA: A DEep learning based Language Technology plAtform","date":"2019-08-02","arxiv_id":"1908.01853","repositories_listed":2,"syntology":null},{"url":"/paper/cif-continuous-integrate-and-fire-for-end-to","slug":"cif-continuous-integrate-and-fire-for-end-to","title":"CIF: Continuous Integrate-and-Fire for End-to-End Speech Recognition","date":"2019-05-27","arxiv_id":"1905.11235","repositories_listed":2,"syntology":null},{"url":"/paper/rwth-asr-systems-for-librispeech-hybrid-vs","slug":"rwth-asr-systems-for-librispeech-hybrid-vs","title":"RWTH ASR Systems for LibriSpeech: Hybrid vs Attention -- w/o Data Augmentation","date":"2019-05-08","arxiv_id":"1905.03072","repositories_listed":2,"syntology":null},{"url":"/paper/mitigating-the-impact-of-speech-recognition-1","slug":"mitigating-the-impact-of-speech-recognition-1","title":"Mitigating the Impact of Speech Recognition Errors on Spoken Question Answering by Adversarial Domain Adaptation","date":"2019-04-16","arxiv_id":"1904.07904","repositories_listed":2,"syntology":null},{"url":"/paper/speech-and-speaker-recognition-from-raw","slug":"speech-and-speaker-recognition-from-raw","title":"Speech and Speaker Recognition from Raw Waveform with SincNet","date":"2018-12-13","arxiv_id":"1812.05920","repositories_listed":2,"syntology":null},{"url":"/paper/streaming-end-to-end-speech-recognition-for","slug":"streaming-end-to-end-speech-recognition-for","title":"Streaming End-to-end Speech Recognition For Mobile Devices","date":"2018-11-15","arxiv_id":"1811.06621","repositories_listed":2,"syntology":null},{"url":"/paper/bidirectional-quaternion-long-short-term","slug":"bidirectional-quaternion-long-short-term","title":"Bidirectional Quaternion Long-Short Term Memory Recurrent Neural Networks for Speech Recognition","date":"2018-11-06","arxiv_id":"1811.02566","repositories_listed":2,"syntology":null},{"url":"/paper/how2-a-large-scale-dataset-for-multimodal","slug":"how2-a-large-scale-dataset-for-multimodal","title":"How2: A Large-scale Dataset for Multimodal Language Understanding","date":"2018-11-01","arxiv_id":"1811.00347","repositories_listed":2,"syntology":null},{"url":"/paper/generative-adversarial-networks-for-unpaired","slug":"generative-adversarial-networks-for-unpaired","title":"Generative Adversarial Networks for Unpaired Voice Transformation on Impaired Speech","date":"2018-10-30","arxiv_id":"1810.12656","repositories_listed":2,"syntology":null},{"url":"/paper/lrw-1000-a-naturally-distributed-large-scale","slug":"lrw-1000-a-naturally-distributed-large-scale","title":"LRW-1000: A Naturally-Distributed Large-Scale Benchmark for Lip Reading in the Wild","date":"2018-10-16","arxiv_id":"1810.06990","repositories_listed":2,"syntology":null},{"url":"/paper/optimal-completion-distillation-for-sequence","slug":"optimal-completion-distillation-for-sequence","title":"Optimal Completion Distillation for Sequence Learning","date":"2018-10-02","arxiv_id":"1810.01398","repositories_listed":2,"syntology":null},{"url":"/paper/open-source-automatic-speech-recognition-for","slug":"open-source-automatic-speech-recognition-for","title":"Open Source Automatic Speech Recognition for German","date":"2018-07-26","arxiv_id":"1807.10311","repositories_listed":2,"syntology":null},{"url":"/paper/augmented-cyclic-adversarial-learning-for-low","slug":"augmented-cyclic-adversarial-learning-for-low","title":"Augmented Cyclic Adversarial Learning for Low Resource Domain Adaptation","date":"2018-07-01","arxiv_id":"1807.00374","repositories_listed":2,"syntology":null},{"url":"/paper/twin-regularization-for-online-speech","slug":"twin-regularization-for-online-speech","title":"Twin Regularization for online speech recognition","date":"2018-04-15","arxiv_id":"1804.05374","repositories_listed":2,"syntology":null},{"url":"/paper/scalable-factorized-hierarchical-variational","slug":"scalable-factorized-hierarchical-variational","title":"Scalable Factorized Hierarchical Variational Autoencoder Training","date":"2018-04-09","arxiv_id":"1804.03201","repositories_listed":2,"syntology":null},{"url":"/paper/long-short-term-memory-and-learning-to-learn","slug":"long-short-term-memory-and-learning-to-learn","title":"Long short-term memory and learning-to-learn in networks of spiking neurons","date":"2018-03-26","arxiv_id":"1803.09574","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/long-short-term-memory-and-learning-to-learn#ran","syntology_url":"https://syntology.ai/paper/1803.09574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.09574"}},"official":null}},{"url":"/paper/end-to-end-audiovisual-speech-recognition","slug":"end-to-end-audiovisual-speech-recognition","title":"End-to-end Audiovisual Speech Recognition","date":"2018-02-18","arxiv_id":"1802.06424","repositories_listed":2,"syntology":null},{"url":"/paper/training-rnns-as-fast-as-cnns","slug":"training-rnns-as-fast-as-cnns","title":"Training RNNs as Fast as CNNs","date":"2018-01-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/letter-based-speech-recognition-with-gated","slug":"letter-based-speech-recognition-with-gated","title":"Letter-Based Speech Recognition with Gated ConvNets","date":"2017-12-22","arxiv_id":"1712.09444","repositories_listed":2,"syntology":null},{"url":"/paper/minimum-word-error-rate-training-for","slug":"minimum-word-error-rate-training-for","title":"Minimum Word Error Rate Training for Attention-based Sequence-to-Sequence Models","date":"2017-12-05","arxiv_id":"1712.01818","repositories_listed":2,"syntology":null},{"url":"/paper/attention-based-models-for-text-dependent","slug":"attention-based-models-for-text-dependent","title":"Attention-Based Models for Text-Dependent Speaker Verification","date":"2017-10-28","arxiv_id":"1710.10470","repositories_listed":2,"syntology":null},{"url":"/paper/rotational-unit-of-memory","slug":"rotational-unit-of-memory","title":"Rotational Unit of Memory","date":"2017-10-26","arxiv_id":"1710.09537","repositories_listed":2,"syntology":null},{"url":"/paper/the-dirha-english-corpus-and-related-tasks","slug":"the-dirha-english-corpus-and-related-tasks","title":"The DIRHA-English corpus and related tasks for distant-speech recognition in domestic environments","date":"2017-10-06","arxiv_id":"1710.02560","repositories_listed":2,"syntology":null},{"url":"/paper/aishell-1-an-open-source-mandarin-speech","slug":"aishell-1-an-open-source-mandarin-speech","title":"AISHELL-1: An Open-Source Mandarin Speech Corpus and A Speech Recognition Baseline","date":"2017-09-16","arxiv_id":"1709.05522","repositories_listed":2,"syntology":null},{"url":"/paper/twin-networks-matching-the-future-for","slug":"twin-networks-matching-the-future-for","title":"Twin Networks: Matching the Future for Sequence Generation","date":"2017-08-22","arxiv_id":"1708.06742","repositories_listed":2,"syntology":null},{"url":"/paper/chemception-a-deep-neural-network-with","slug":"chemception-a-deep-neural-network-with","title":"Chemception: A Deep Neural Network with Minimal Chemistry Knowledge Matches the Performance of Expert-developed QSAR/QSPR Models","date":"2017-06-20","arxiv_id":"1706.06689","repositories_listed":2,"syntology":null},{"url":"/paper/3d-convolutional-neural-networks-for-cross","slug":"3d-convolutional-neural-networks-for-cross","title":"3D Convolutional Neural Networks for Cross Audio-Visual Matching Recognition","date":"2017-06-18","arxiv_id":"1706.05739","repositories_listed":2,"syntology":null},{"url":"/paper/online-and-linear-time-attention-by-enforcing","slug":"online-and-linear-time-attention-by-enforcing","title":"Online and Linear-Time Attention by Enforcing Monotonic Alignments","date":"2017-04-03","arxiv_id":"1704.00784","repositories_listed":2,"syntology":null},{"url":"/paper/sequence-modeling-via-segmentations","slug":"sequence-modeling-via-segmentations","title":"Sequence Modeling via Segmentations","date":"2017-02-24","arxiv_id":"1702.07463","repositories_listed":2,"syntology":null},{"url":"/paper/regularizing-neural-networks-by-penalizing","slug":"regularizing-neural-networks-by-penalizing","title":"Regularizing Neural Networks by Penalizing Confident Output Distributions","date":"2017-01-23","arxiv_id":"1701.06548","repositories_listed":2,"syntology":null},{"url":"/paper/very-deep-convolutional-networks-for-end-to","slug":"very-deep-convolutional-networks-for-end-to","title":"Very Deep Convolutional Networks for End-to-End Speech Recognition","date":"2016-10-10","arxiv_id":"1610.03022","repositories_listed":2,"syntology":null},{"url":"/paper/qsgd-communication-efficient-sgd-via-gradient","slug":"qsgd-communication-efficient-sgd-via-gradient","title":"QSGD: Communication-Efficient SGD via Gradient Quantization and Encoding","date":"2016-10-07","arxiv_id":"1610.02132","repositories_listed":2,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/qsgd-communication-efficient-sgd-via-gradient#ran","syntology_url":"https://syntology.ai/paper/1610.02132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1610.02132"}},"official":null}},{"url":"/paper/very-deep-convolutional-neural-networks-for-1","slug":"very-deep-convolutional-neural-networks-for-1","title":"Very Deep Convolutional Neural Networks for Robust Speech Recognition","date":"2016-10-02","arxiv_id":"1610.00277","repositories_listed":2,"syntology":null},{"url":"/paper/trainable-frontend-for-robust-and-far-field","slug":"trainable-frontend-for-robust-and-far-field","title":"Trainable Frontend For Robust and Far-Field Keyword Spotting","date":"2016-07-19","arxiv_id":"1607.05666","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/trainable-frontend-for-robust-and-far-field#ran","syntology_url":"https://syntology.ai/paper/1607.05666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1607.05666"}},"official":null}},{"url":"/paper/dsd-dense-sparse-dense-training-for-deep","slug":"dsd-dense-sparse-dense-training-for-deep","title":"DSD: Dense-Sparse-Dense Training for Deep Neural Networks","date":"2016-07-15","arxiv_id":"1607.04381","repositories_listed":2,"syntology":null},{"url":"/paper/single-channel-multi-speaker-separation-using","slug":"single-channel-multi-speaker-separation-using","title":"Single-Channel Multi-Speaker Separation using Deep Clustering","date":"2016-07-07","arxiv_id":"1607.02173","repositories_listed":2,"syntology":null},{"url":"/paper/using-filter-banks-in-convolutional-neural","slug":"using-filter-banks-in-convolutional-neural","title":"Using Filter Banks in Convolutional Neural Networks for Texture Classification","date":"2016-01-12","arxiv_id":"1601.02919","repositories_listed":2,"syntology":null},{"url":"/paper/strategies-for-training-large-vocabulary","slug":"strategies-for-training-large-vocabulary","title":"Strategies for Training Large Vocabulary Neural Language Models","date":"2015-12-15","arxiv_id":"1512.04906","repositories_listed":2,"syntology":null},{"url":"/paper/an-end-to-end-neural-network-for-polyphonic","slug":"an-end-to-end-neural-network-for-polyphonic","title":"An End-to-End Neural Network for Polyphonic Piano Music Transcription","date":"2015-08-07","arxiv_id":"1508.01774","repositories_listed":2,"syntology":null},{"url":"/paper/do-deep-nets-really-need-to-be-deep","slug":"do-deep-nets-really-need-to-be-deep","title":"Do Deep Nets Really Need to be Deep?","date":"2013-12-21","arxiv_id":"1312.6184","repositories_listed":2,"syntology":null},{"url":"/paper/connectionist-temporal-classification","slug":"connectionist-temporal-classification","title":"Connectionist Temporal Classification: Labelling Unsegmented Sequence Data with Recurrent Neural Networks","date":"2006-07-25","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/first-steps-towards-voice-anonymization-for","slug":"first-steps-towards-voice-anonymization-for","title":"First Steps Towards Voice Anonymization for Code-Switching Speech","date":"2025-07-02","arxiv_id":"2507.01765","repositories_listed":1,"syntology":null},{"url":"/paper/splitformer-an-improved-early-exit","slug":"splitformer-an-improved-early-exit","title":"Splitformer: An improved early-exit architecture for automatic speech recognition on edge devices","date":"2025-06-22","arxiv_id":"2506.18035","repositories_listed":1,"syntology":null},{"url":"/paper/advances-in-small-footprint-keyword-spotting","slug":"advances-in-small-footprint-keyword-spotting","title":"Advances in Small-Footprint Keyword Spotting: A Comprehensive Review of Efficient Models and Algorithms","date":"2025-06-12","arxiv_id":"2506.11169","repositories_listed":1,"syntology":null},{"url":"/paper/addressing-pitfalls-in-auditing-practices-of","slug":"addressing-pitfalls-in-auditing-practices-of","title":"Addressing Pitfalls in Auditing Practices of Automatic Speech Recognition Technologies: A Case Study of People with Aphasia","date":"2025-06-10","arxiv_id":"2506.08846","repositories_listed":1,"syntology":null},{"url":"/paper/structured-state-space-model-dynamics-and","slug":"structured-state-space-model-dynamics-and","title":"Structured State Space Model Dynamics and Parametrization for Spiking Neural Networks","date":"2025-06-04","arxiv_id":"2506.06374","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-rhythm-and-voice-conversion-to","slug":"unsupervised-rhythm-and-voice-conversion-to","title":"Unsupervised Rhythm and Voice Conversion to Improve ASR on Dysarthric Speech","date":"2025-06-02","arxiv_id":"2506.01618","repositories_listed":1,"syntology":null},{"url":"/paper/gigaam-efficient-self-supervised-learner-for","slug":"gigaam-efficient-self-supervised-learner-for","title":"GigaAM: Efficient Self-Supervised Learner for Speech Recognition","date":"2025-06-01","arxiv_id":"2506.01192","repositories_listed":1,"syntology":null},{"url":"/paper/what-do-self-supervised-speech-models-know-1","slug":"what-do-self-supervised-speech-models-know-1","title":"What do self-supervised speech models know about Dutch? Analyzing advantages of language-specific pre-training","date":"2025-06-01","arxiv_id":"2506.00981","repositories_listed":1,"syntology":null},{"url":"/paper/towards-temporally-explainable-dysarthric","slug":"towards-temporally-explainable-dysarthric","title":"Towards Temporally Explainable Dysarthric Speech Clarity Assessment","date":"2025-05-31","arxiv_id":"2506.00454","repositories_listed":1,"syntology":null},{"url":"/paper/vedavani-a-benchmark-corpus-for-asr-on-vedic","slug":"vedavani-a-benchmark-corpus-for-asr-on-vedic","title":"Vedavani: A Benchmark Corpus for ASR on Vedic Sanskrit Poetry","date":"2025-05-30","arxiv_id":"2506.00145","repositories_listed":1,"syntology":null},{"url":"/paper/beavertalk-oregon-state-university-s-iwslt","slug":"beavertalk-oregon-state-university-s-iwslt","title":"BeaverTalk: Oregon State University's IWSLT 2025 Simultaneous Speech Translation System","date":"2025-05-29","arxiv_id":"2505.24016","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-generative-error-correction-for","slug":"exploring-generative-error-correction-for","title":"Exploring Generative Error Correction for Dysarthric Speech Recognition","date":"2025-05-26","arxiv_id":"2505.20163","repositories_listed":1,"syntology":null},{"url":"/paper/chser-a-dataset-and-case-study-on-generative","slug":"chser-a-dataset-and-case-study-on-generative","title":"CHSER: A Dataset and Case Study on Generative Speech Error Correction for Child ASR","date":"2025-05-24","arxiv_id":"2505.18463","repositories_listed":1,"syntology":null},{"url":"/paper/daily-omni-towards-audio-visual-reasoning","slug":"daily-omni-towards-audio-visual-reasoning","title":"Daily-Omni: Towards Audio-Visual Reasoning with Temporal Alignment across Modalities","date":"2025-05-23","arxiv_id":"2505.17862","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/daily-omni-towards-audio-visual-reasoning#ran","syntology_url":"https://syntology.ai/paper/2505.17862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17862"}},"official":{"repos":["lliar-liar/daily-omni"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/llm-based-generative-error-correction-for","slug":"llm-based-generative-error-correction-for","title":"LLM-based Generative Error Correction for Rare Words with Synthetic Data and Phonetic Context","date":"2025-05-23","arxiv_id":"2505.17410","repositories_listed":1,"syntology":null},{"url":"/paper/speechless-speech-instruction-training","slug":"speechless-speech-instruction-training","title":"Speechless: Speech Instruction Training Without Speech for Low Resource Languages","date":"2025-05-23","arxiv_id":"2505.17417","repositories_listed":1,"syntology":null},{"url":"/paper/from-tens-of-hours-to-tens-of-thousands","slug":"from-tens-of-hours-to-tens-of-thousands","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","date":"2025-05-22","arxiv_id":"2505.16972","repositories_listed":1,"syntology":null},{"url":"/paper/word-level-timestamp-generation-for-automatic","slug":"word-level-timestamp-generation-for-automatic","title":"Word Level Timestamp Generation for Automatic Speech Recognition and Translation","date":"2025-05-21","arxiv_id":"2505.15646","repositories_listed":1,"syntology":null},{"url":"/paper/dual-precision-quantization-for-efficient-and","slug":"dual-precision-quantization-for-efficient-and","title":"Dual Precision Quantization for Efficient and Accurate Deep Neural Networks Inference","date":"2025-05-20","arxiv_id":"2505.14638","repositories_listed":1,"syntology":null},{"url":"/paper/personatab-predicting-personality-traits","slug":"personatab-predicting-personality-traits","title":"PersonaTAB: Predicting Personality Traits using Textual, Acoustic, and Behavioral Cues in Fully-Duplex Speech Dialogs","date":"2025-05-20","arxiv_id":"2505.14356","repositories_listed":1,"syntology":null},{"url":"/paper/towards-inclusive-asr-investigating-voice","slug":"towards-inclusive-asr-investigating-voice","title":"Towards Inclusive ASR: Investigating Voice Conversion for Dysarthric Speech Recognition in Low-Resource Languages","date":"2025-05-20","arxiv_id":"2505.14874","repositories_listed":1,"syntology":null},{"url":"/paper/transfer-learning-from-visual-speech","slug":"transfer-learning-from-visual-speech","title":"Transfer Learning from Visual Speech Recognition to Mouthing Recognition in German Sign Language","date":"2025-05-20","arxiv_id":"2505.13784","repositories_listed":1,"syntology":null},{"url":"/paper/multi-head-temporal-latent-attention","slug":"multi-head-temporal-latent-attention","title":"Multi-head Temporal Latent Attention","date":"2025-05-19","arxiv_id":"2505.13544","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-head-temporal-latent-attention#ran","syntology_url":"https://syntology.ai/paper/2505.13544","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13544"}},"official":{"repos":["d-keqi/mlta"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-10879","slug":"2505-10879","title":"Multi-Stage Speaker Diarization for Noisy Classrooms","date":"2025-05-16","arxiv_id":"2505.10879","repositories_listed":1,"syntology":null},{"url":"/paper/cogenav-versatile-audio-visual-representation","slug":"cogenav-versatile-audio-visual-representation","title":"CoGenAV: Versatile Audio-Visual Representation Learning via Contrastive-Generative Synchronization","date":"2025-05-06","arxiv_id":"2505.03186","repositories_listed":1,"syntology":null},{"url":"/paper/vita-audio-fast-interleaved-cross-modal-token","slug":"vita-audio-fast-interleaved-cross-modal-token","title":"VITA-Audio: Fast Interleaved Cross-Modal Token Generation for Efficient Large Speech-Language Model","date":"2025-05-06","arxiv_id":"2505.03739","repositories_listed":1,"syntology":null},{"url":"/paper/voila-voice-language-foundation-models-for","slug":"voila-voice-language-foundation-models-for","title":"Voila: Voice-Language Foundation Models for Real-Time Autonomous Interaction and Voice Role-Play","date":"2025-05-05","arxiv_id":"2505.02707","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/voila-voice-language-foundation-models-for#ran","syntology_url":"https://syntology.ai/paper/2505.02707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02707"}},"official":{"repos":["maitrix-org/voila"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/transforming-faces-into-video-stories","slug":"transforming-faces-into-video-stories","title":"Transforming faces into video stories -- VideoFace2.0","date":"2025-05-04","arxiv_id":"2505.02060","repositories_listed":1,"syntology":null},{"url":"/paper/bersting-at-the-screams-a-benchmark-for","slug":"bersting-at-the-screams-a-benchmark-for","title":"BERSting at the Screams: A Benchmark for Distanced, Emotional and Shouted Speech Recognition","date":"2025-04-30","arxiv_id":"2505.00059","repositories_listed":1,"syntology":null},{"url":"/paper/kimi-audio-technical-report","slug":"kimi-audio-technical-report","title":"Kimi-Audio Technical Report","date":"2025-04-25","arxiv_id":"2504.18425","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/kimi-audio-technical-report#ran","syntology_url":"https://syntology.ai/paper/2504.18425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.18425"}},"official":{"repos":["moonshotai/kimi-audio"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tinyml-for-speech-recognition","slug":"tinyml-for-speech-recognition","title":"TinyML for Speech Recognition","date":"2025-04-22","arxiv_id":"2504.16213","repositories_listed":1,"syntology":null},{"url":"/paper/dysarthria-normalization-via-local-lie-group","slug":"dysarthria-normalization-via-local-lie-group","title":"Dysarthria Normalization via Local Lie Group Transformations for Robust ASR","date":"2025-04-16","arxiv_id":"2504.12279","repositories_listed":1,"syntology":null},{"url":"/paper/rnn-transducer-based-losses-for-speech","slug":"rnn-transducer-based-losses-for-speech","title":"RNN-Transducer-based Losses for Speech Recognition on Noisy Targets","date":"2025-04-09","arxiv_id":"2504.06963","repositories_listed":1,"syntology":null}],"record_sha256":"15f7191ca0cbc36091478ec596f9d3be315058d14ec4bc67867fb70e7a27dbe3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}