{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/14","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":14,"pages_in_order":65,"rows_per_page":100,"rows":[1301,1400],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/13","next":"/task/speech-recognition/papers/15","papers":[{"url":"/paper/deep-fsmn-for-large-vocabulary-continuous","slug":"deep-fsmn-for-large-vocabulary-continuous","title":"Deep-FSMN for Large Vocabulary Continuous Speech Recognition","date":"2018-03-04","arxiv_id":"1803.05030","repositories_listed":1,"syntology":null},{"url":"/paper/xnmt-the-extensible-neural-machine","slug":"xnmt-the-extensible-neural-machine","title":"XNMT: The eXtensible Neural Machine Translation Toolkit","date":"2018-03-01","arxiv_id":"1803.00188","repositories_listed":1,"syntology":null},{"url":"/paper/from-gameplay-to-symbolic-reasoning-learning","slug":"from-gameplay-to-symbolic-reasoning-learning","title":"From Gameplay to Symbolic Reasoning: Learning SAT Solver Heuristics in the Style of Alpha(Go) Zero","date":"2018-02-14","arxiv_id":"1802.05340","repositories_listed":1,"syntology":null},{"url":"/paper/augmenting-librispeech-with-french","slug":"augmenting-librispeech-with-french","title":"Augmenting Librispeech with French Translations: A Multimodal Corpus for Direct Speech Translation Evaluation","date":"2018-02-09","arxiv_id":"1802.03142","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-neural-network-based-semantic","slug":"recurrent-neural-network-based-semantic","title":"Recurrent Neural Network-Based Semantic Variational Autoencoder for Sequence-to-Sequence Learning","date":"2018-02-09","arxiv_id":"1802.03238","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-past-mistakes-improving","slug":"learning-from-past-mistakes-improving","title":"Learning from Past Mistakes: Improving Automatic Speech Recognition Output via Noisy-Clean Phrase Context Modeling","date":"2018-02-07","arxiv_id":"1802.02607","repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-a-critical-appraisal","slug":"deep-learning-a-critical-appraisal","title":"Deep Learning: A Critical Appraisal","date":"2018-01-02","arxiv_id":"1801.00631","repositories_listed":1,"syntology":null},{"url":"/paper/did-you-hear-that-adversarial-examples","slug":"did-you-hear-that-adversarial-examples","title":"Did you hear that? Adversarial Examples Against Automatic Speech Recognition","date":"2018-01-02","arxiv_id":"1801.00554","repositories_listed":1,"syntology":null},{"url":"/paper/monotonic-chunkwise-attention-1","slug":"monotonic-chunkwise-attention-1","title":"Monotonic Chunkwise Attention","date":"2018-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/use-of-deep-learning-in-modern-recommendation","slug":"use-of-deep-learning-in-modern-recommendation","title":"Use of Deep Learning in Modern Recommendation System: A Summary of Recent Works","date":"2017-12-20","arxiv_id":"1712.07525","repositories_listed":1,"syntology":null},{"url":"/paper/monotonic-chunkwise-attention","slug":"monotonic-chunkwise-attention","title":"Monotonic Chunkwise Attention","date":"2017-12-14","arxiv_id":"1712.05382","repositories_listed":1,"syntology":null},{"url":"/paper/using-rule-based-labels-for-weak-supervised","slug":"using-rule-based-labels-for-weak-supervised","title":"Using Rule-Based Labels for Weak Supervised Learning: A ChemNet for Transferable Chemical Property Prediction","date":"2017-12-07","arxiv_id":"1712.02734","repositories_listed":1,"syntology":null},{"url":"/paper/phonemic-transcription-of-low-resource-tonal","slug":"phonemic-transcription-of-low-resource-tonal","title":"Phonemic Transcription of Low-Resource Tonal Languages","date":"2017-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/action-recognition-in-video-sequences-using","slug":"action-recognition-in-video-sequences-using","title":"Action Recognition in Video Sequences using Deep Bi-Directional LSTM With CNN Features","date":"2017-11-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/language-bootstrapping-learning-word-meanings","slug":"language-bootstrapping-learning-word-meanings","title":"Language Bootstrapping: Learning Word Meanings From Perception-Action Association","date":"2017-11-27","arxiv_id":"1711.09714","repositories_listed":1,"syntology":null},{"url":"/paper/improved-training-for-online-end-to-end","slug":"improved-training-for-online-end-to-end","title":"Improved training for online end-to-end speech recognition systems","date":"2017-11-06","arxiv_id":"1711.02212","repositories_listed":1,"syntology":null},{"url":"/paper/deep-word-embeddings-for-visual-speech","slug":"deep-word-embeddings-for-visual-speech","title":"Deep word embeddings for visual speech recognition","date":"2017-10-30","arxiv_id":"1710.11201","repositories_listed":1,"syntology":null},{"url":"/paper/the-implementation-of-a-deep-recurrent-neural","slug":"the-implementation-of-a-deep-recurrent-neural","title":"The implementation of a Deep Recurrent Neural Network Language Model on a Xilinx FPGA","date":"2017-10-26","arxiv_id":"1710.10296","repositories_listed":1,"syntology":null},{"url":"/paper/trace-norm-regularization-and-faster","slug":"trace-norm-regularization-and-faster","title":"Trace norm regularization and faster inference for embedded speech recognition RNNs","date":"2017-10-25","arxiv_id":"1710.09026","repositories_listed":1,"syntology":null},{"url":"/paper/benchmark-of-deep-learning-models-on-large","slug":"benchmark-of-deep-learning-models-on-large","title":"Benchmark of Deep Learning Models on Large Healthcare MIMIC Datasets","date":"2017-10-23","arxiv_id":"1710.08531","repositories_listed":1,"syntology":null},{"url":"/paper/contaminated-speech-training-methods-for","slug":"contaminated-speech-training-methods-for","title":"Contaminated speech training methods for robust DNN-HMM distant speech recognition","date":"2017-10-10","arxiv_id":"1710.03538","repositories_listed":1,"syntology":null},{"url":"/paper/improving-speech-recognition-by-revising","slug":"improving-speech-recognition-by-revising","title":"Improving speech recognition by revising gated recurrent units","date":"2017-09-29","arxiv_id":"1710.00641","repositories_listed":1,"syntology":null},{"url":"/paper/speech-recognition-challenge-in-the-wild","slug":"speech-recognition-challenge-in-the-wild","title":"Speech Recognition Challenge in the Wild: Arabic MGB-3","date":"2017-09-21","arxiv_id":"1709.07276","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-hidden-representations-in-end-to","slug":"analyzing-hidden-representations-in-end-to","title":"Analyzing Hidden Representations in End-to-End Automatic Speech Recognition Systems","date":"2017-09-13","arxiv_id":"1709.04482","repositories_listed":1,"syntology":null},{"url":"/paper/spoken-english-intelligibility-remediation","slug":"spoken-english-intelligibility-remediation","title":"Spoken English Intelligibility Remediation with PocketSphinx Alignment and Feature Extraction Improves Substantially over the State of the Art","date":"2017-09-06","arxiv_id":"1709.01713","repositories_listed":1,"syntology":null},{"url":"/paper/language-identification-using-deep","slug":"language-identification-using-deep","title":"Language Identification Using Deep Convolutional Recurrent Neural Networks","date":"2017-08-16","arxiv_id":"1708.04811","repositories_listed":1,"syntology":null},{"url":"/paper/massively-multilingual-neural-grapheme-to","slug":"massively-multilingual-neural-grapheme-to","title":"Massively Multilingual Neural Grapheme-to-Phoneme Conversion","date":"2017-08-04","arxiv_id":"1708.01464","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-submodular-rank-aggregation-on","slug":"unsupervised-submodular-rank-aggregation-on","title":"Unsupervised Submodular Rank Aggregation on Score-based Permutations","date":"2017-07-04","arxiv_id":"1707.01166","repositories_listed":1,"syntology":null},{"url":"/paper/dual-supervised-learning","slug":"dual-supervised-learning","title":"Dual Supervised Learning","date":"2017-07-03","arxiv_id":"1707.00415","repositories_listed":1,"syntology":null},{"url":"/paper/improving-lstm-ctc-based-asr-performance-in","slug":"improving-lstm-ctc-based-asr-performance-in","title":"Improving LSTM-CTC based ASR performance in domains with limited training data","date":"2017-07-03","arxiv_id":"1707.00722","repositories_listed":1,"syntology":null},{"url":"/paper/joint-ctcattention-decoding-for-end-to-end","slug":"joint-ctcattention-decoding-for-end-to-end","title":"Joint CTC/attention decoding for end-to-end speech recognition","date":"2017-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rank-1-constrained-multichannel-wiener-filter","slug":"rank-1-constrained-multichannel-wiener-filter","title":"Rank-1 Constrained Multichannel Wiener Filter for Speech Recognition in Noisy Environments","date":"2017-07-01","arxiv_id":"1707.00201","repositories_listed":1,"syntology":null},{"url":"/paper/fast-slow-recurrent-neural-networks","slug":"fast-slow-recurrent-neural-networks","title":"Fast-Slow Recurrent Neural Networks","date":"2017-05-24","arxiv_id":"1705.08639","repositories_listed":1,"syntology":null},{"url":"/paper/a-design-methodology-for-efficient","slug":"a-design-methodology-for-efficient","title":"A Design Methodology for Efficient Implementation of Deconvolutional Neural Networks on an FPGA","date":"2017-05-07","arxiv_id":"1705.02583","repositories_listed":1,"syntology":null},{"url":"/paper/speech-based-visual-question-answering","slug":"speech-based-visual-question-answering","title":"Speech-Based Visual Question Answering","date":"2017-05-01","arxiv_id":"1705.00464","repositories_listed":1,"syntology":null},{"url":"/paper/robust-training-under-linguistic-adversity","slug":"robust-training-under-linguistic-adversity","title":"Robust Training under Linguistic Adversity","date":"2017-04-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sequence-to-sequence-models-can-directly","slug":"sequence-to-sequence-models-can-directly","title":"Sequence-to-Sequence Models Can Directly Translate Foreign Speech","date":"2017-03-24","arxiv_id":"1703.08581","repositories_listed":1,"syntology":null},{"url":"/paper/a-morphology-aware-network-for-morphological","slug":"a-morphology-aware-network-for-morphological","title":"A Morphology-aware Network for Morphological Disambiguation","date":"2017-02-13","arxiv_id":"1702.03654","repositories_listed":1,"syntology":null},{"url":"/paper/first-automatic-fongbe-continuous-speech","slug":"first-automatic-fongbe-continuous-speech","title":"First Automatic Fongbe Continuous Speech Recognition System: Development of Acoustic Models and Language Models","date":"2017-01-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towards-end-to-end-speech-recognition-with","slug":"towards-end-to-end-speech-recognition-with","title":"Towards End-to-End Speech Recognition with Deep Convolutional Neural Networks","date":"2017-01-10","arxiv_id":"1701.02720","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/towards-end-to-end-speech-recognition-with#ran","syntology_url":"https://syntology.ai/paper/1701.02720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1701.02720"}},"official":null}},{"url":"/paper/deep-learning-and-its-applications-to-machine","slug":"deep-learning-and-its-applications-to-machine","title":"Deep Learning and Its Applications to Machine Health Monitoring: A Survey","date":"2016-12-16","arxiv_id":"1612.07640","repositories_listed":1,"syntology":null},{"url":"/paper/audio-segmentation-for-robust-real-time","slug":"audio-segmentation-for-robust-real-time","title":"Audio Segmentation for Robust Real-Time Speech Recognition Based on Neural Networks","date":"2016-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/combination-of-convolutional-and-recurrent","slug":"combination-of-convolutional-and-recurrent","title":"Combination of Convolutional and Recurrent Neural Network for Sentiment Analysis of Short Texts","date":"2016-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/convolutional-neural-network-language-models","slug":"convolutional-neural-network-language-models","title":"Convolutional Neural Network Language Models","date":"2016-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/latent-tree-language-model-1","slug":"latent-tree-language-model-1","title":"Latent Tree Language Model","date":"2016-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/neural-sentiment-classification-with-user-and","slug":"neural-sentiment-classification-with-user-and","title":"Neural Sentiment Classification with User and Product Attention","date":"2016-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-gentle-tutorial-of-recurrent-neural-network","slug":"a-gentle-tutorial-of-recurrent-neural-network","title":"A Gentle Tutorial of Recurrent Neural Network with Error Backpropagation","date":"2016-10-08","arxiv_id":"1610.02583","repositories_listed":1,"syntology":null},{"url":"/paper/collecting-resources-in-sub-saharan-african","slug":"collecting-resources-in-sub-saharan-african","title":"Collecting Resources in Sub-Saharan African Languages for Automatic Speech Recognition: a Case Study of Wolof","date":"2016-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towards-bayesian-deep-learning-a-survey","slug":"towards-bayesian-deep-learning-a-survey","title":"A Survey on Bayesian Deep Learning","date":"2016-04-06","arxiv_id":"1604.01662","repositories_listed":1,"syntology":null},{"url":"/paper/character-level-neural-translation-for","slug":"character-level-neural-translation-for","title":"Character-Level Neural Translation for Multilingual Media Monitoring in the SUMMA Project","date":"2016-04-05","arxiv_id":"1604.01221","repositories_listed":1,"syntology":null},{"url":"/paper/neural-network-based-spectral-mask-estimation","slug":"neural-network-based-spectral-mask-estimation","title":"Neural network based spectral mask estimation for acoustic beamforming","date":"2016-03-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/character-level-incremental-speech","slug":"character-level-incremental-speech","title":"Character-Level Incremental Speech Recognition with Recurrent Neural Networks","date":"2016-01-25","arxiv_id":"1601.06581","repositories_listed":1,"syntology":null},{"url":"/paper/open-source-german-distant-speech-recognition","slug":"open-source-german-distant-speech-recognition","title":"Open Source German Distant Speech Recognition: Corpus and Acoustic Model","date":"2015-12-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/thchs-30-a-free-chinese-speech-corpus","slug":"thchs-30-a-free-chinese-speech-corpus","title":"THCHS-30 : A Free Chinese Speech Corpus","date":"2015-12-07","arxiv_id":"1512.01882","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/thchs-30-a-free-chinese-speech-corpus#ran","syntology_url":"https://syntology.ai/paper/1512.01882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1512.01882"}},"official":null}},{"url":"/paper/calibrated-structured-prediction","slug":"calibrated-structured-prediction","title":"Calibrated Structured Prediction","date":"2015-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/task-loss-estimation-for-sequence-prediction","slug":"task-loss-estimation-for-sequence-prediction","title":"Task Loss Estimation for Sequence Prediction","date":"2015-11-19","arxiv_id":"1511.06456","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/task-loss-estimation-for-sequence-prediction#ran","syntology_url":"https://syntology.ai/paper/1511.06456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1511.06456"}},"official":{"repos":["rizar/attention-lvcsr"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/deep-convolutional-acoustic-word-embeddings","slug":"deep-convolutional-acoustic-word-embeddings","title":"Deep convolutional acoustic word embeddings using word-pair side information","date":"2015-10-05","arxiv_id":"1510.01032","repositories_listed":1,"syntology":null},{"url":"/paper/high-order-graph-based-neural-dependency","slug":"high-order-graph-based-neural-dependency","title":"High-order Graph-based Neural Dependency Parsing","date":"2015-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/automatic-dialect-detection-in-arabic","slug":"automatic-dialect-detection-in-arabic","title":"Automatic Dialect Detection in Arabic Broadcast Speech","date":"2015-09-23","arxiv_id":"1509.06928","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-visualization-of-what-a-deep","slug":"evaluating-the-visualization-of-what-a-deep","title":"Evaluating the visualization of what a Deep Neural Network has learned","date":"2015-09-21","arxiv_id":"1509.06321","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-attention-based-large-vocabulary","slug":"end-to-end-attention-based-large-vocabulary","title":"End-to-End Attention-based Large Vocabulary Speech Recognition","date":"2015-08-18","arxiv_id":"1508.04395","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-word-embedding-for-contrasting","slug":"revisiting-word-embedding-for-contrasting","title":"Revisiting Word Embedding for Contrasting Meaning","date":"2015-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/beyond-temporal-pooling-recurrence-and","slug":"beyond-temporal-pooling-recurrence-and","title":"Beyond Temporal Pooling: Recurrence and Temporal Convolutions for Gesture Recognition in Video","date":"2015-06-05","arxiv_id":"1506.01911","repositories_listed":1,"syntology":null},{"url":"/paper/convolutional-neural-network-for-paraphrase","slug":"convolutional-neural-network-for-paraphrase","title":"Convolutional Neural Network for Paraphrase Identification","date":"2015-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-probabilistic-theory-of-deep-learning","slug":"a-probabilistic-theory-of-deep-learning","title":"A Probabilistic Theory of Deep Learning","date":"2015-04-02","arxiv_id":"1504.00641","repositories_listed":1,"syntology":null},{"url":"/paper/parallel-training-of-dnns-with-natural","slug":"parallel-training-of-dnns-with-natural","title":"Parallel training of DNNs with Natural Gradient and Parameter Averaging","date":"2014-10-27","arxiv_id":"1410.7455","repositories_listed":1,"syntology":null},{"url":"/paper/building-dnn-acoustic-models-for-large","slug":"building-dnn-acoustic-models-for-large","title":"Building DNN Acoustic Models for Large Vocabulary Speech Recognition","date":"2014-06-30","arxiv_id":"1406.7806","repositories_listed":1,"syntology":null},{"url":"/paper/a-generalized-language-model-as-the-1","slug":"a-generalized-language-model-as-the-1","title":"A Generalized Language Model as the Combination of Skipped n-grams and Modified Kneser Ney Smoothing","date":"2014-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/lex4all-a-language-independent-tool-for","slug":"lex4all-a-language-independent-tool-for","title":"lex4all: A language-independent tool for building and evaluating pronunciation lexicons for small-vocabulary speech recognition","date":"2014-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-toolkit-for-efficient-learning-of-lexical","slug":"a-toolkit-for-efficient-learning-of-lexical","title":"A Toolkit for Efficient Learning of Lexical Units for Speech Recognition","date":"2014-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/long-short-term-memory-based-recurrent-neural","slug":"long-short-term-memory-based-recurrent-neural","title":"Long Short-Term Memory Based Recurrent Neural Network Architectures for Large Vocabulary Speech Recognition","date":"2014-02-05","arxiv_id":"1402.1128","repositories_listed":1,"syntology":null},{"url":"/paper/learning-human-pose-estimation-features-with","slug":"learning-human-pose-estimation-features-with","title":"Learning Human Pose Estimation Features with Convolutional Networks","date":"2013-12-27","arxiv_id":"1312.7302","repositories_listed":1,"syntology":null},{"url":"/paper/keyphrase-cloud-generation-of-broadcast-news","slug":"keyphrase-cloud-generation-of-broadcast-news","title":"Keyphrase Cloud Generation of Broadcast News","date":"2013-06-19","arxiv_id":"1306.4606","repositories_listed":1,"syntology":null},{"url":null,"slug":"nonverbaltts-a-public-english-corpus-of-text","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","date":"2025-07-17","arxiv_id":"2507.13155","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-specific-audio-coding-for-machines","title":"Task-Specific Audio Coding for Machines: Machine-Learned Latent Features Are Codes for That Machine","date":"2025-07-17","arxiv_id":"2507.12701","repositories_listed":0,"syntology":null},{"url":null,"slug":"whisperkit-on-device-real-time-asr-with","title":"WhisperKit: On-device Real-time ASR with Billion-Scale Transformers","date":"2025-07-14","arxiv_id":"2507.10860","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualspeaker-visually-guided-3d-avatar-lip","title":"VisualSpeaker: Visually-Guided 3D Avatar Lip Synthesis","date":"2025-07-08","arxiv_id":"2507.06060","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hybrid-machine-learning-framework-for","title":"A Hybrid Machine Learning Framework for Optimizing Crop Selection via Agronomic and Economic Forecasting","date":"2025-07-06","arxiv_id":"2507.08832","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-pronunciation-mistake-detector","title":"AUTOMATIC PRONUNCIATION MISTAKE DETECTOR PROJECT REPORT","date":"2025-06-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-target-speaker-based-overlap","title":"Lightweight Target-Speaker-Based Overlap Transcription for Practical Streaming ASR","date":"2025-06-25","arxiv_id":"2506.20288","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-representation-learning-and-fusion","title":"Multimodal Representation Learning and Fusion","date":"2025-06-25","arxiv_id":"2506.20494","repositories_listed":0,"syntology":null},{"url":null,"slug":"voice-control-robot-using-arduino-management","title":"VOICE CONTROL ROBOT USING ARDUINO MANAGEMENT SYSTEM PROJECT.","date":"2025-06-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-generated-song-detection-via-lyrics","title":"AI-Generated Song Detection via Lyrics Transcripts","date":"2025-06-23","arxiv_id":"2506.18488","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-spoken-grammatical-error","title":"End-to-End Spoken Grammatical Error Correction","date":"2025-06-23","arxiv_id":"2506.18532","repositories_listed":0,"syntology":null},{"url":null,"slug":"opuslm-a-family-of-open-unified-speech","title":"OpusLM: A Family of Open Unified Speech Language Models","date":"2025-06-21","arxiv_id":"2506.17611","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-the-transcription-bottleneck-fine","title":"Breaking the Transcription Bottleneck: Fine-tuning ASR Models for Extremely Low-Resource Fieldwork Languages","date":"2025-06-20","arxiv_id":"2506.17459","repositories_listed":0,"syntology":null},{"url":null,"slug":"lm-spt-lm-aligned-semantic-distillation-for","title":"LM-SPT: LM-Aligned Semantic Distillation for Speech Tokenization","date":"2025-06-20","arxiv_id":"2506.16738","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-space-models-in-efficient-whispered-and","title":"State-Space Models in Efficient Whispered and Multi-dialect Speech Recognition","date":"2025-06-20","arxiv_id":"2506.16969","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-speech-recognition-biases-in","title":"Automatic Speech Recognition Biases in Newcastle English: an Error Analysis","date":"2025-06-19","arxiv_id":"2506.16558","repositories_listed":0,"syntology":null},{"url":null,"slug":"weight-factorization-and-centralization-for","title":"Weight Factorization and Centralization for Continual Learning in Speech Recognition","date":"2025-06-19","arxiv_id":"2506.16574","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-practical-aspects-of-end-to-end","title":"Improving Practical Aspects of End-to-End Multi-Talker Speech Recognition for Online and Offline Scenarios","date":"2025-06-17","arxiv_id":"2506.14204","repositories_listed":0,"syntology":null},{"url":null,"slug":"thinking-in-directivity-speech-large-language","title":"Thinking in Directivity: Speech Large Language Model for Multi-Talker Directional Speech Recognition","date":"2025-06-17","arxiv_id":"2506.14973","repositories_listed":0,"syntology":null},{"url":"/paper/unifying-streaming-and-non-streaming","slug":"unifying-streaming-and-non-streaming","title":"Unifying Streaming and Non-streaming Zipformer-based ASR","date":"2025-06-17","arxiv_id":"2506.14434","repositories_listed":0,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unifying-streaming-and-non-streaming#ran","syntology_url":"https://syntology.ai/paper/2506.14434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.14434"}},"official":null}},{"url":null,"slug":"a-silent-speech-decoding-system-from-eeg-and","title":"A Silent Speech Decoding System from EEG and EMG with Heterogenous Electrode Configurations","date":"2025-06-16","arxiv_id":"2506.13835","repositories_listed":0,"syntology":null},{"url":null,"slug":"bi-directional-context-enhanced-speech-large","title":"Bi-directional Context-Enhanced Speech Large Language Models for Multilingual Conversational ASR","date":"2025-06-16","arxiv_id":"2506.13396","repositories_listed":0,"syntology":null},{"url":null,"slug":"but-system-for-the-mlc-slm-challenge","title":"BUT System for the MLC-SLM Challenge","date":"2025-06-16","arxiv_id":"2506.13414","repositories_listed":0,"syntology":null},{"url":null,"slug":"ntu-speechlab-llm-based-multilingual-asr","title":"NTU Speechlab LLM-Based Multilingual ASR System for Interspeech MLC-SLM Challenge 2025","date":"2025-06-16","arxiv_id":"2506.13339","repositories_listed":0,"syntology":null},{"url":null,"slug":"qwen-vs-gemma-integration-with-whisper-a","title":"Qwen vs. Gemma Integration with Whisper: A Comparative Study in Multilingual SpeechLLM Systems","date":"2025-06-16","arxiv_id":"2506.13596","repositories_listed":0,"syntology":null},{"url":null,"slug":"seewo-s-submission-to-mlc-slm-lessons-learned","title":"Seewo's Submission to MLC-SLM: Lessons learned from Speech Reasoning Language Models","date":"2025-06-16","arxiv_id":"2506.13300","repositories_listed":0,"syntology":null},{"url":null,"slug":"sc-sot-conditioning-the-decoder-on-diarized","title":"SC-SOT: Conditioning the Decoder on Diarized Speaker Information for End-to-End Overlapped Speech Recognition","date":"2025-06-15","arxiv_id":"2506.12672","repositories_listed":0,"syntology":null}],"record_sha256":"950775c3609a1cd07969e8c08b26d51da5cc6aa4ab7fd322bec0d8fe883a3bc1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}