{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/11","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":65,"rows_per_page":100,"rows":[1001,1100],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/10","next":"/task/speech-recognition/papers/12","papers":[{"url":"/paper/using-radio-archives-for-low-resource-speech","slug":"using-radio-archives-for-low-resource-speech","title":"Using Radio Archives for Low-Resource Speech Recognition: Towards an Intelligent Virtual Assistant for Illiterate Users","date":"2021-04-27","arxiv_id":"2104.13083","repositories_listed":1,"syntology":null},{"url":"/paper/lebenchmark-a-reproducible-framework-for","slug":"lebenchmark-a-reproducible-framework-for","title":"LeBenchmark: A Reproducible Framework for Assessing Self-Supervised Representation Learning from Speech","date":"2021-04-23","arxiv_id":"2104.11462","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-and-generic-1d-dilated-convolution","slug":"efficient-and-generic-1d-dilated-convolution","title":"Efficient and Generic 1D Dilated Convolution Layer for Deep Learning","date":"2021-04-16","arxiv_id":"2104.08002","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-keyword-spotting-through-long-range","slug":"efficient-keyword-spotting-through-long-range","title":"Efficient Keyword Spotting by capturing long-range interactions with Temporal Lambda Networks","date":"2021-04-16","arxiv_id":"2104.08086","repositories_listed":1,"syntology":null},{"url":"/paper/a-method-to-reveal-speaker-identity-in","slug":"a-method-to-reveal-speaker-identity-in","title":"A Method to Reveal Speaker Identity in Distributed ASR Training, and How to Counter It","date":"2021-04-15","arxiv_id":"2104.07815","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-independence-for-pretext-task","slug":"conditional-independence-for-pretext-task","title":"Conditional independence for pretext task selection in Self-supervised speech representation learning","date":"2021-04-15","arxiv_id":"2104.07388","repositories_listed":1,"syntology":null},{"url":"/paper/cross-domain-speech-recognition-with","slug":"cross-domain-speech-recognition-with","title":"Cross-domain Speech Recognition with Unsupervised Character-level Distribution Matching","date":"2021-04-15","arxiv_id":"2104.07491","repositories_listed":1,"syntology":null},{"url":"/paper/eat-enhanced-asr-tts-for-self-supervised","slug":"eat-enhanced-asr-tts-for-self-supervised","title":"EAT: Enhanced ASR-TTS for Self-supervised Speech Recognition","date":"2021-04-13","arxiv_id":"2104.07474","repositories_listed":1,"syntology":null},{"url":"/paper/nemo-inverse-text-normalization-from","slug":"nemo-inverse-text-normalization-from","title":"NeMo Inverse Text Normalization: From Development To Production","date":"2021-04-11","arxiv_id":"2104.05055","repositories_listed":1,"syntology":null},{"url":"/paper/nemo-toolbox-for-speech-dataset-construction","slug":"nemo-toolbox-for-speech-dataset-construction","title":"A Toolbox for Construction and Analysis of Speech Datasets","date":"2021-04-11","arxiv_id":"2104.04896","repositories_listed":1,"syntology":null},{"url":"/paper/rnn-transducer-models-for-spoken-language","slug":"rnn-transducer-models-for-spoken-language","title":"RNN Transducer Models For Spoken Language Understanding","date":"2021-04-08","arxiv_id":"2104.03842","repositories_listed":1,"syntology":null},{"url":"/paper/speak-or-chat-with-me-end-to-end-spoken","slug":"speak-or-chat-with-me-end-to-end-spoken","title":"Speak or Chat with Me: End-to-End Spoken Language Understanding System with Flexible Inputs","date":"2021-04-07","arxiv_id":"2104.05752","repositories_listed":1,"syntology":null},{"url":"/paper/ai4d-african-language-program","slug":"ai4d-african-language-program","title":"AI4D -- African Language Program","date":"2021-04-06","arxiv_id":"2104.02516","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-rank-microphones-for-distant","slug":"learning-to-rank-microphones-for-distant","title":"Learning to Rank Microphones for Distant Speech Recognition","date":"2021-04-06","arxiv_id":"2104.02819","repositories_listed":1,"syntology":null},{"url":"/paper/lt-lm-a-novel-non-autoregressive-language","slug":"lt-lm-a-novel-non-autoregressive-language","title":"LT-LM: a novel non-autoregressive language model for single-shot lattice rescoring","date":"2021-04-06","arxiv_id":"2104.02526","repositories_listed":1,"syntology":null},{"url":"/paper/spgispeech-5000-hours-of-transcribed","slug":"spgispeech-5000-hours-of-transcribed","title":"SPGISpeech: 5,000 hours of transcribed financial audio for fully formatted end-to-end speech recognition","date":"2021-04-05","arxiv_id":"2104.02014","repositories_listed":1,"syntology":null},{"url":"/paper/tsnat-two-step-non-autoregressvie-transformer","slug":"tsnat-two-step-non-autoregressvie-transformer","title":"TSNAT: Two-Step Non-Autoregressvie Transformer Models for Speech Recognition","date":"2021-04-04","arxiv_id":"2104.01522","repositories_listed":1,"syntology":null},{"url":"/paper/exkaldi-rt-a-real-time-automatic-speech","slug":"exkaldi-rt-a-real-time-automatic-speech","title":"ExKaldi-RT: A Real-Time Automatic Speech Recognition Extension Toolkit of Kaldi","date":"2021-04-03","arxiv_id":"2104.01384","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-fly-aligned-data-augmentation-for","slug":"on-the-fly-aligned-data-augmentation-for","title":"On-the-Fly Aligned Data Augmentation for Sequence-to-Sequence ASR","date":"2021-04-03","arxiv_id":"2104.01393","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-and-code-switching-asr","slug":"multilingual-and-code-switching-asr","title":"Multilingual and code-switching ASR challenges for low resource Indian languages","date":"2021-04-01","arxiv_id":"2104.00235","repositories_listed":1,"syntology":null},{"url":"/paper/q-asr-integer-only-zero-shot-quantization-for","slug":"q-asr-integer-only-zero-shot-quantization-for","title":"Integer-only Zero-shot Quantization for Efficient Speech Recognition","date":"2021-03-31","arxiv_id":"2103.16827","repositories_listed":1,"syntology":null},{"url":"/paper/mediaspeech-multilanguage-asr-benchmark-and","slug":"mediaspeech-multilanguage-asr-benchmark-and","title":"MediaSpeech: Multilanguage ASR Benchmark and Dataset","date":"2021-03-30","arxiv_id":"2103.16193","repositories_listed":1,"syntology":null},{"url":"/paper/libri-adhoc40-a-dataset-collected-from","slug":"libri-adhoc40-a-dataset-collected-from","title":"Libri-adhoc40: A dataset collected from synchronized ad-hoc microphone arrays","date":"2021-03-28","arxiv_id":"2103.15118","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-bias-in-automatic-speech","slug":"quantifying-bias-in-automatic-speech","title":"Quantifying Bias in Automatic Speech Recognition","date":"2021-03-28","arxiv_id":"2103.15122","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-neural-representations-for","slug":"leveraging-neural-representations-for","title":"Leveraging pre-trained representations to improve access to untranscribed speech from endangered languages","date":"2021-03-26","arxiv_id":"2103.14583","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-low-resource-phoneme-recognition-on","slug":"real-time-low-resource-phoneme-recognition-on","title":"Real-time low-resource phoneme recognition on edge devices","date":"2021-03-25","arxiv_id":"2103.13997","repositories_listed":1,"syntology":null},{"url":"/paper/sok-a-modularized-approach-to-study-the","slug":"sok-a-modularized-approach-to-study-the","title":"SoK: A Modularized Approach to Study the Security of Automatic Speech Recognition Systems","date":"2021-03-19","arxiv_id":"2103.10651","repositories_listed":1,"syntology":null},{"url":"/paper/fast-development-of-asr-in-african-languages","slug":"fast-development-of-asr-in-african-languages","title":"Fast Development of ASR in African Languages using Self Supervised Speech Representation Learning","date":"2021-03-16","arxiv_id":"2103.08993","repositories_listed":1,"syntology":null},{"url":"/paper/a-parallelizable-lattice-rescoring-strategy","slug":"a-parallelizable-lattice-rescoring-strategy","title":"A Parallelizable Lattice Rescoring Strategy with Neural Language Models","date":"2021-03-08","arxiv_id":"2103.05081","repositories_listed":1,"syntology":null},{"url":"/paper/waveguard-understanding-and-mitigating-audio","slug":"waveguard-understanding-and-mitigating-audio","title":"WaveGuard: Understanding and Mitigating Audio Adversarial Examples","date":"2021-03-04","arxiv_id":"2103.03344","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/waveguard-understanding-and-mitigating-audio#ran","syntology_url":"https://syntology.ai/paper/2103.03344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.03344"}},"official":null}},{"url":"/paper/exploiting-attention-based-sequence-to","slug":"exploiting-attention-based-sequence-to","title":"Exploiting Attention-based Sequence-to-Sequence Architectures for Sound Event Localization","date":"2021-02-28","arxiv_id":"2103.00417","repositories_listed":1,"syntology":null},{"url":"/paper/data-fusion-for-audiovisual-speaker","slug":"data-fusion-for-audiovisual-speaker","title":"Data Fusion for Audiovisual Speaker Localization: Extending Dynamic Stream Weights to the Spatial Domain","date":"2021-02-23","arxiv_id":"2102.11588","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-phonetic-neural-model-for-correction","slug":"hybrid-phonetic-neural-model-for-correction","title":"Hybrid phonetic-neural model for correction in speech recognition systems","date":"2021-02-12","arxiv_id":"2102.06744","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-approaches-for-automatic","slug":"transformer-based-approaches-for-automatic","title":"Transformer-Based Approaches for Automatic Music Transcription","date":"2021-02-12","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/transformer-language-models-with-lstm-based","slug":"transformer-language-models-with-lstm-based","title":"Transformer Language Models with LSTM-based Cross-utterance Information Representation","date":"2021-02-12","arxiv_id":"2102.06474","repositories_listed":1,"syntology":null},{"url":"/paper/an-investigation-of-end-to-end-models-for","slug":"an-investigation-of-end-to-end-models-for","title":"An Investigation of End-to-End Models for Robust Speech Recognition","date":"2021-02-11","arxiv_id":"2102.06237","repositories_listed":1,"syntology":null},{"url":"/paper/dompteur-taming-audio-adversarial-examples","slug":"dompteur-taming-audio-adversarial-examples","title":"Dompteur: Taming Audio Adversarial Examples","date":"2021-02-10","arxiv_id":"2102.05431","repositories_listed":1,"syntology":null},{"url":"/paper/effects-of-layer-freezing-when-transferring","slug":"effects-of-layer-freezing-when-transferring","title":"Effects of Layer Freezing on Transferring a Speech Recognition System to Under-resourced Languages","date":"2021-02-08","arxiv_id":"2102.04097","repositories_listed":1,"syntology":null},{"url":"/paper/confusion2vec-2-0-enriching-ambiguous-spoken","slug":"confusion2vec-2-0-enriching-ambiguous-spoken","title":"Confusion2vec 2.0: Enriching Ambiguous Spoken Language Representations with Subwords","date":"2021-02-03","arxiv_id":"2102.02270","repositories_listed":1,"syntology":null},{"url":"/paper/bendr-using-transformers-and-a-contrastive","slug":"bendr-using-transformers-and-a-contrastive","title":"BENDR: using transformers and a contrastive self-supervised learning task to learn from massive amounts of EEG data","date":"2021-01-28","arxiv_id":"2101.12037","repositories_listed":1,"syntology":null},{"url":"/paper/arabic-speech-recognition-by-end-to-end","slug":"arabic-speech-recognition-by-end-to-end","title":"Arabic Speech Recognition by End-to-End, Modular Systems and Human","date":"2021-01-21","arxiv_id":"2101.08454","repositories_listed":1,"syntology":null},{"url":"/paper/learning-efficient-representations-for-3","slug":"learning-efficient-representations-for-3","title":"Learning Efficient Representations for Keyword Spotting with Triplet Loss","date":"2021-01-12","arxiv_id":"2101.04792","repositories_listed":1,"syntology":null},{"url":"/paper/data-quality-measures-and-efficient","slug":"data-quality-measures-and-efficient","title":"Data Quality Measures and Efficient Evaluation Algorithms for Large-Scale High-Dimensional Data","date":"2021-01-05","arxiv_id":"2101.01441","repositories_listed":1,"syntology":null},{"url":"/paper/devi-open-source-human-robot-interface-for","slug":"devi-open-source-human-robot-interface-for","title":"DEVI: Open-source Human-Robot Interface for Interactive Receptionist Systems","date":"2021-01-02","arxiv_id":"2101.00479","repositories_listed":1,"syntology":null},{"url":"/paper/voxpopuli-a-large-scale-multilingual-speech","slug":"voxpopuli-a-large-scale-multilingual-speech","title":"VoxPopuli: A Large-Scale Multilingual Speech Corpus for Representation Learning, Semi-Supervised Learning and Interpretation","date":"2021-01-02","arxiv_id":"2101.00390","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voxpopuli-a-large-scale-multilingual-speech#ran","syntology_url":"https://syntology.ai/paper/2101.00390","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.00390"}},"official":{"repos":["facebookresearch/voxpopuli"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-optical-learning-operator","slug":"scalable-optical-learning-operator","title":"Scalable Optical Learning Operator","date":"2020-12-22","arxiv_id":"2012.12404","repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-transformers-as-energy-based-1","slug":"pre-training-transformers-as-energy-based-1","title":"Pre-Training Transformers as Energy-Based Cloze Models","date":"2020-12-15","arxiv_id":"2012.08561","repositories_listed":1,"syntology":null},{"url":"/paper/av-taris-online-audio-visual-speech","slug":"av-taris-online-audio-visual-speech","title":"AV Taris: Online Audio-Visual Speech Recognition","date":"2020-12-14","arxiv_id":"2012.07467","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-learning-for-deep-neural-network","slug":"bayesian-learning-for-deep-neural-network","title":"Bayesian Learning for Deep Neural Network Adaptation","date":"2020-12-14","arxiv_id":"2012.07460","repositories_listed":1,"syntology":null},{"url":"/paper/lips-don-t-lie-a-generalisable-and-robust","slug":"lips-don-t-lie-a-generalisable-and-robust","title":"Lips Don't Lie: A Generalisable and Robust Approach to Face Forgery Detection","date":"2020-12-14","arxiv_id":"2012.07657","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lips-don-t-lie-a-generalisable-and-robust#ran","syntology_url":"https://syntology.ai/paper/2012.07657","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.07657"}},"official":{"repos":["ahaliassos/lipforensics"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/decoar-2-0-deep-contextualized-acoustic","slug":"decoar-2-0-deep-contextualized-acoustic","title":"DeCoAR 2.0: Deep Contextualized Acoustic Representations with Vector Quantization","date":"2020-12-11","arxiv_id":"2012.06659","repositories_listed":1,"syntology":null},{"url":"/paper/on-knowledge-distillation-for-direct-speech","slug":"on-knowledge-distillation-for-direct-speech","title":"On Knowledge Distillation for Direct Speech Translation","date":"2020-12-09","arxiv_id":"2012.04964","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-disentanglement-of-speaker","slug":"adversarial-disentanglement-of-speaker","title":"Adversarial Disentanglement of Speaker Representation for Attribute-Driven Privacy Preservation","date":"2020-12-08","arxiv_id":"2012.04454","repositories_listed":1,"syntology":null},{"url":"/paper/mls-a-large-scale-multilingual-dataset-for","slug":"mls-a-large-scale-multilingual-dataset-for","title":"MLS: A Large-Scale Multilingual Dataset for Speech Research","date":"2020-12-07","arxiv_id":"2012.03411","repositories_listed":1,"syntology":null},{"url":"/paper/parallel-blockwise-knowledge-distillation-for","slug":"parallel-blockwise-knowledge-distillation-for","title":"Parallel Blockwise Knowledge Distillation for Deep Neural Network Compression","date":"2020-12-05","arxiv_id":"2012.03096","repositories_listed":1,"syntology":null},{"url":"/paper/speakingfaces-a-large-scale-multimodal","slug":"speakingfaces-a-large-scale-multimodal","title":"SpeakingFaces: A Large-Scale Multimodal Dataset of Voice Commands with Visual and Thermal Video Streams","date":"2020-12-05","arxiv_id":"2012.02961","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-asr-system-with-automatic","slug":"end-to-end-asr-system-with-automatic","title":"End to End ASR System with Automatic Punctuation Insertion","date":"2020-12-03","arxiv_id":"2012.02012","repositories_listed":1,"syntology":null},{"url":"/paper/attentively-embracing-noise-for-robust-latent","slug":"attentively-embracing-noise-for-robust-latent","title":"Attentively Embracing Noise for Robust Latent Representation in BERT","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-automatic-speech-recognition-for","slug":"end-to-end-automatic-speech-recognition-for","title":"End-to-End Automatic Speech Recognition for Gujarati","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/metacat-a-metadata-based-task-oriented","slug":"metacat-a-metadata-based-task-oriented","title":"metaCAT: A Metadata-based Task-oriented Chatbot Annotation Tool","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sar-net-a-end-to-end-deep-speech-accent","slug":"sar-net-a-end-to-end-deep-speech-accent","title":"Deep Discriminative Feature Learning for Accent Recognition","date":"2020-11-25","arxiv_id":"2011.12461","repositories_listed":1,"syntology":null},{"url":"/paper/an-online-multilingual-hate-speech","slug":"an-online-multilingual-hate-speech","title":"An Online Multilingual Hate speech Recognition System","date":"2020-11-23","arxiv_id":"2011.11523","repositories_listed":1,"syntology":null},{"url":"/paper/wpd-an-improved-neural-beamformer-for","slug":"wpd-an-improved-neural-beamformer-for","title":"WPD++: An Improved Neural Beamformer for Simultaneous Speech Separation and Dereverberation","date":"2020-11-18","arxiv_id":"2011.09162","repositories_listed":1,"syntology":null},{"url":"/paper/deep-compressive-offloading-speeding-up","slug":"deep-compressive-offloading-speeding-up","title":"Deep Compressive Offloading: Speeding Up Neural Network Inference by Trading Edge Computation for Network Latency","date":"2020-11-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learn-an-effective-lip-reading-model-without","slug":"learn-an-effective-lip-reading-model-without","title":"Learn an Effective Lip Reading Model without Pains","date":"2020-11-15","arxiv_id":"2011.07557","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-neural-architecture-search-for-end","slug":"efficient-neural-architecture-search-for-end","title":"Efficient Neural Architecture Search for End-to-end Speech Recognition via Straight-Through Gradients","date":"2020-11-11","arxiv_id":"2011.05649","repositories_listed":1,"syntology":null},{"url":"/paper/text-augmentation-for-language-models-in-high","slug":"text-augmentation-for-language-models-in-high","title":"Text Augmentation for Language Models in High Error Recognition Scenario","date":"2020-11-11","arxiv_id":"2011.06056","repositories_listed":1,"syntology":null},{"url":"/paper/nanopore-base-calling-on-the-edge","slug":"nanopore-base-calling-on-the-edge","title":"Nanopore Base Calling on the Edge","date":"2020-11-09","arxiv_id":"2011.04312","repositories_listed":1,"syntology":null},{"url":"/paper/domain-adaptation-using-class-similarity-for","slug":"domain-adaptation-using-class-similarity-for","title":"Domain Adaptation Using Class Similarity for Robust Speech Recognition","date":"2020-11-05","arxiv_id":"2011.02782","repositories_listed":1,"syntology":null},{"url":"/paper/improving-rnn-transducer-based-asr-with","slug":"improving-rnn-transducer-based-asr-with","title":"Improving RNN Transducer Based ASR with Auxiliary Tasks","date":"2020-11-05","arxiv_id":"2011.03109","repositories_listed":1,"syntology":null},{"url":"/paper/dnn-based-mask-estimation-for-distributed","slug":"dnn-based-mask-estimation-for-distributed","title":"DNN-based mask estimation for distributed speech enhancement in spatially unconstrained microphone arrays","date":"2020-11-03","arxiv_id":"2011.01714","repositories_listed":1,"syntology":null},{"url":"/paper/minimum-bayes-risk-training-for-end-to-end","slug":"minimum-bayes-risk-training-for-end-to-end","title":"Minimum Bayes Risk Training for End-to-End Speaker-Attributed ASR","date":"2020-11-03","arxiv_id":"2011.02921","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-pretrained-transformer-to-lattices","slug":"adapting-pretrained-transformer-to-lattices","title":"Adapting Pretrained Transformer to Lattices for Spoken Language Understanding","date":"2020-11-02","arxiv_id":"2011.00780","repositories_listed":1,"syntology":null},{"url":"/paper/dual-decoder-transformer-for-joint-automatic","slug":"dual-decoder-transformer-for-joint-automatic","title":"Dual-decoder Transformer for Joint Automatic Speech Recognition and Multilingual Speech Translation","date":"2020-11-02","arxiv_id":"2011.00747","repositories_listed":1,"syntology":null},{"url":"/paper/direct-segmentation-models-for-streaming","slug":"direct-segmentation-models-for-streaming","title":"Direct Segmentation Models for Streaming Speech Translation","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/punctuation-restoration-using-transformer","slug":"punctuation-restoration-using-transformer","title":"Punctuation Restoration using Transformer Models for High-and Low-Resource Languages","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-bottleneck-features-for","slug":"multilingual-bottleneck-features-for","title":"Multilingual Bottleneck Features for Improving ASR Performance of Code-Switched Speech in Under-Resourced Languages","date":"2020-10-31","arxiv_id":"2011.03118","repositories_listed":1,"syntology":null},{"url":"/paper/joint-masked-cpc-and-ctc-training-for-asr","slug":"joint-masked-cpc-and-ctc-training-for-asr","title":"Joint Masked CPC and CTC Training for ASR","date":"2020-10-30","arxiv_id":"2011.00093","repositories_listed":1,"syntology":null},{"url":"/paper/speech-simclr-combining-contrastive-and","slug":"speech-simclr-combining-contrastive-and","title":"Speech SIMCLR: Combining Contrastive and Reconstruction Objective for Self-supervised Speech Representation Learning","date":"2020-10-27","arxiv_id":"2010.13991","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/speech-simclr-combining-contrastive-and#ran","syntology_url":"https://syntology.ai/paper/2010.13991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.13991"}},"official":{"repos":["athena-team/athena"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-scale-end-to-end-multilingual-speech","slug":"large-scale-end-to-end-multilingual-speech","title":"Large-Scale End-to-End Multilingual Speech Recognition and Language Identification with Multi-Task Learning","date":"2020-10-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/probing-acoustic-representations-for-phonetic","slug":"probing-acoustic-representations-for-phonetic","title":"Probing Acoustic Representations for Phonetic Properties","date":"2020-10-25","arxiv_id":"2010.13007","repositories_listed":1,"syntology":null},{"url":"/paper/two-stage-textual-knowledge-distillation-to","slug":"two-stage-textual-knowledge-distillation-to","title":"Two-stage Textual Knowledge Distillation for End-to-End Spoken Language Understanding","date":"2020-10-25","arxiv_id":"2010.13105","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-end-to-end-speech","slug":"transformer-based-end-to-end-speech","title":"Transformer-based End-to-End Speech Recognition with Local Dense Synthesizer Attention","date":"2020-10-23","arxiv_id":"2010.12155","repositories_listed":1,"syntology":null},{"url":"/paper/confidence-estimation-for-attention-based","slug":"confidence-estimation-for-attention-based","title":"Confidence Estimation for Attention-based Sequence-to-sequence Models for Speech Recognition","date":"2020-10-22","arxiv_id":"2010.11428","repositories_listed":1,"syntology":null},{"url":"/paper/how-phonotactics-affect-multilingual-and-zero","slug":"how-phonotactics-affect-multilingual-and-zero","title":"How Phonotactics Affect Multilingual and Zero-shot ASR Performance","date":"2020-10-22","arxiv_id":"2010.12104","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-evaluation-in-asr-are-our-models","slug":"rethinking-evaluation-in-asr-are-our-models","title":"Rethinking Evaluation in ASR: Are Our Models Robust Enough?","date":"2020-10-22","arxiv_id":"2010.11745","repositories_listed":1,"syntology":null},{"url":"/paper/emformer-efficient-memory-transformer-based","slug":"emformer-efficient-memory-transformer-based","title":"Emformer: Efficient Memory Transformer Based Acoustic Model For Low Latency Streaming Speech Recognition","date":"2020-10-21","arxiv_id":"2010.10759","repositories_listed":1,"syntology":null},{"url":"/paper/fastemit-low-latency-streaming-asr-with","slug":"fastemit-low-latency-streaming-asr-with","title":"FastEmit: Low-latency Streaming ASR with Sequence-level Emission Regularization","date":"2020-10-21","arxiv_id":"2010.11148","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/fastemit-low-latency-streaming-asr-with#ran","syntology_url":"https://syntology.ai/paper/2010.11148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.11148"}},"official":null}},{"url":"/paper/towards-end-to-end-training-of-automatic","slug":"towards-end-to-end-training-of-automatic","title":"Towards End-to-End Training of Automatic Speech Recognition for Nigerian Pidgin","date":"2020-10-21","arxiv_id":"2010.11123","repositories_listed":1,"syntology":null},{"url":"/paper/venomave-clean-label-poisoning-against-speech","slug":"venomave-clean-label-poisoning-against-speech","title":"VenoMave: Targeted Poisoning Against Speech Recognition","date":"2020-10-21","arxiv_id":"2010.10682","repositories_listed":1,"syntology":null},{"url":"/paper/pushing-the-limits-of-semi-supervised","slug":"pushing-the-limits-of-semi-supervised","title":"Pushing the Limits of Semi-Supervised Learning for Automatic Speech Recognition","date":"2020-10-20","arxiv_id":"2010.10504","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pushing-the-limits-of-semi-supervised#ran","syntology_url":"https://syntology.ai/paper/2010.10504","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10504"}},"official":null}},{"url":"/paper/google-crowdsourced-speech-corpora-and","slug":"google-crowdsourced-speech-corpora-and","title":"Google Crowdsourced Speech Corpora and Related Open-Source Resources for Low-Resource Languages and Dialects: An Overview","date":"2020-10-14","arxiv_id":"2010.06778","repositories_listed":1,"syntology":null},{"url":"/paper/swiss-parliaments-corpus-an-automatically","slug":"swiss-parliaments-corpus-an-automatically","title":"Swiss Parliaments Corpus, an Automatically Aligned Swiss German Speech to Standard German Text Corpus","date":"2020-10-06","arxiv_id":"2010.02810","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-grounding-for-multimodal-speech","slug":"fine-grained-grounding-for-multimodal-speech","title":"Fine-Grained Grounding for Multimodal Speech Recognition","date":"2020-10-05","arxiv_id":"2010.02384","repositories_listed":1,"syntology":null},{"url":"/paper/online-neural-networks-for-change-point","slug":"online-neural-networks-for-change-point","title":"Online Neural Networks for Change-Point Detection","date":"2020-10-03","arxiv_id":"2010.01388","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-weighted-finite-state","slug":"differentiable-weighted-finite-state","title":"Differentiable Weighted Finite-State Transducers","date":"2020-10-02","arxiv_id":"2010.01003","repositories_listed":1,"syntology":null},{"url":"/paper/improving-vietnamese-named-entity-recognition","slug":"improving-vietnamese-named-entity-recognition","title":"Improving Vietnamese Named Entity Recognition from Speech Using Word Capitalization and Punctuation Recovery Models","date":"2020-10-01","arxiv_id":"2010.00198","repositories_listed":1,"syntology":null},{"url":"/paper/a-crowdsourced-open-source-kazakh-speech","slug":"a-crowdsourced-open-source-kazakh-speech","title":"A Crowdsourced Open-Source Kazakh Speech Corpus and Initial Speech Recognition Baseline","date":"2020-09-22","arxiv_id":"2009.10334","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-learning-of-speech-2d-feature","slug":"end-to-end-learning-of-speech-2d-feature","title":"End-to-End Learning of Speech 2D Feature-Trajectory for Prosthetic Hands","date":"2020-09-22","arxiv_id":"2009.10283","repositories_listed":1,"syntology":null},{"url":"/paper/sdst-successive-decoding-for-speech-to-text","slug":"sdst-successive-decoding-for-speech-to-text","title":"Consecutive Decoding for Speech-to-text Translation","date":"2020-09-21","arxiv_id":"2009.09737","repositories_listed":1,"syntology":null}],"record_sha256":"358e84382b8edd12677f69f66f3acc47c2cb0e674dbe5a8a919e1e2f5e2334db","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}