{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition/papers/13","list_of":"/task/speech-recognition","task":"Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":13,"pages_in_order":65,"rows_per_page":100,"rows":[1201,1300],"of":6433,"counts":{"archive_papers_tagged":6433,"with_a_code_link":1373,"where_syntology_ran_a_sample":196,"not_listed_spam_title":0,"listed":6433,"listed_where_code_ran":196,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":162,"every_run_a_failure_of_syntologys_instrument":34,"listed_with_a_run_with_no_instrument_failure":162,"listed_every_run_a_failure_of_syntologys_instrument":34,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition","prev":"/task/speech-recognition/papers/12","next":"/task/speech-recognition/papers/14","papers":[{"url":"/paper/adversarial-example-detection-by","slug":"adversarial-example-detection-by","title":"Adversarial Example Detection by Classification for Deep Speech Recognition","date":"2019-10-22","arxiv_id":"1910.10013","repositories_listed":1,"syntology":null},{"url":"/paper/gpu-accelerated-viterbi-exact-lattice-decoder","slug":"gpu-accelerated-viterbi-exact-lattice-decoder","title":"GPU-Accelerated Viterbi Exact Lattice Decoder for Batched Online and Offline Speech Recognition","date":"2019-10-22","arxiv_id":"1910.10032","repositories_listed":1,"syntology":null},{"url":"/paper/improving-transformer-based-speech","slug":"improving-transformer-based-speech","title":"Improving Transformer-based Speech Recognition Using Unsupervised Pre-training","date":"2019-10-22","arxiv_id":"1910.09932","repositories_listed":1,"syntology":null},{"url":"/paper/indian-emospeech-command-dataset-a-dataset","slug":"indian-emospeech-command-dataset-a-dataset","title":"Indian EmoSpeech Command Dataset: A dataset for emotion based speech recognition in the wild","date":"2019-10-18","arxiv_id":"1910.13801","repositories_listed":1,"syntology":null},{"url":"/paper/librivoxdeen-a-corpus-for-german-to-english","slug":"librivoxdeen-a-corpus-for-german-to-english","title":"LibriVoxDeEn: A Corpus for German-to-English Speech Translation and German Speech Recognition","date":"2019-10-17","arxiv_id":"1910.07924","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-end-to-end-speech-translation","slug":"multilingual-end-to-end-speech-translation","title":"Multilingual End-to-End Speech Translation","date":"2019-10-01","arxiv_id":"1910.00254","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multilingual-end-to-end-speech-translation#ran","syntology_url":"https://syntology.ai/paper/1910.00254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.00254"}},"official":{"repos":["espnet/espnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/state-of-the-art-speech-recognition-using","slug":"state-of-the-art-speech-recognition-using","title":"State-of-the-Art Speech Recognition Using Multi-Stream Self-Attention With Dilated 1D Convolutions","date":"2019-10-01","arxiv_id":"1910.00716","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/state-of-the-art-speech-recognition-using#ran","syntology_url":"https://syntology.ai/paper/1910.00716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.00716"}},"official":null}},{"url":"/paper/fasnet-low-latency-adaptive-beamforming-for","slug":"fasnet-low-latency-adaptive-beamforming-for","title":"FaSNet: Low-latency Adaptive Beamforming for Multi-microphone Audio Processing","date":"2019-09-29","arxiv_id":"1909.13387","repositories_listed":1,"syntology":null},{"url":"/paper/language-agnostic-syllabification-with-neural","slug":"language-agnostic-syllabification-with-neural","title":"Language-Agnostic Syllabification with Neural Sequence Labeling","date":"2019-09-29","arxiv_id":"1909.13362","repositories_listed":1,"syntology":null},{"url":"/paper/improving-rnn-transducer-modeling-for-end-to","slug":"improving-rnn-transducer-modeling-for-end-to","title":"Improving RNN Transducer Modeling for End-to-End Speech Recognition","date":"2019-09-26","arxiv_id":"1909.12415","repositories_listed":1,"syntology":null},{"url":"/paper/disentangling-speech-and-non-speech","slug":"disentangling-speech-and-non-speech","title":"Disentangling Speech and Non-Speech Components for Building Robust Acoustic Models from Found Data","date":"2019-09-25","arxiv_id":"1909.11727","repositories_listed":1,"syntology":null},{"url":"/paper/espresso-a-fast-end-to-end-neural-speech","slug":"espresso-a-fast-end-to-end-neural-speech","title":"Espresso: A Fast End-to-end Neural Speech Recognition Toolkit","date":"2019-09-18","arxiv_id":"1909.08723","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/espresso-a-fast-end-to-end-neural-speech#ran","syntology_url":"https://syntology.ai/paper/1909.08723","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.08723"}},"official":{"repos":["freewym/espresso"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/nemo-a-toolkit-for-building-ai-applications","slug":"nemo-a-toolkit-for-building-ai-applications","title":"NeMo: a toolkit for building AI applications using Neural Modules","date":"2019-09-14","arxiv_id":"1909.09577","repositories_listed":1,"syntology":null},{"url":"/paper/language-learning-using-speech-to-image","slug":"language-learning-using-speech-to-image","title":"Language learning using Speech to Image retrieval","date":"2019-09-09","arxiv_id":"1909.03795","repositories_listed":1,"syntology":null},{"url":"/paper/learning-alignment-for-multimodal-emotion","slug":"learning-alignment-for-multimodal-emotion","title":"Learning Alignment for Multimodal Emotion Recognition from Speech","date":"2019-09-06","arxiv_id":"1909.05645","repositories_listed":1,"syntology":null},{"url":"/paper/bandwidth-embeddings-for-mixed-bandwidth","slug":"bandwidth-embeddings-for-mixed-bandwidth","title":"Bandwidth Embeddings for Mixed-bandwidth Speech Recognition","date":"2019-09-05","arxiv_id":"1909.02667","repositories_listed":1,"syntology":null},{"url":"/paper/two-pass-end-to-end-speech-recognition","slug":"two-pass-end-to-end-speech-recognition","title":"Two-Pass End-to-End Speech Recognition","date":"2019-08-29","arxiv_id":"1908.10992","repositories_listed":1,"syntology":null},{"url":"/paper/emotionless-privacy-preserving-speech","slug":"emotionless-privacy-preserving-speech","title":"Emotionless: Privacy-Preserving Speech Analysis for Voice Assistants","date":"2019-08-09","arxiv_id":"1908.03632","repositories_listed":1,"syntology":null},{"url":"/paper/random-directional-attack-for-fooling-deep","slug":"random-directional-attack-for-fooling-deep","title":"Random Directional Attack for Fooling Deep Neural Networks","date":"2019-08-06","arxiv_id":"1908.02658","repositories_listed":1,"syntology":null},{"url":"/paper/personalizing-asr-for-dysarthric-and-accented","slug":"personalizing-asr-for-dysarthric-and-accented","title":"Personalizing ASR for Dysarthric and Accented Speech with Limited Data","date":"2019-07-31","arxiv_id":"1907.13511","repositories_listed":1,"syntology":null},{"url":"/paper/mass-a-large-and-clean-multilingual-corpus-of","slug":"mass-a-large-and-clean-multilingual-corpus-of","title":"MaSS: A Large and Clean Multilingual Corpus of Sentence-aligned Spoken Utterances Extracted from the Bible","date":"2019-07-30","arxiv_id":"1907.12895","repositories_listed":1,"syntology":null},{"url":"/paper/radiotalk-a-large-scale-corpus-of-talk-radio","slug":"radiotalk-a-large-scale-corpus-of-talk-radio","title":"RadioTalk: a large-scale corpus of talk radio transcripts","date":"2019-07-16","arxiv_id":"1907.07073","repositories_listed":1,"syntology":null},{"url":"/paper/pykaldi2-yet-another-speech-toolkit-based-on","slug":"pykaldi2-yet-another-speech-toolkit-based-on","title":"PyKaldi2: Yet another speech toolkit based on Kaldi and PyTorch","date":"2019-07-12","arxiv_id":"1907.05955","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-phonetic-and-graphemic","slug":"analyzing-phonetic-and-graphemic","title":"Analyzing Phonetic and Graphemic Representations in End-to-End Automatic Speech Recognition","date":"2019-07-09","arxiv_id":"1907.04224","repositories_listed":1,"syntology":null},{"url":"/paper/niesr-nuisance-invariant-end-to-end-speech","slug":"niesr-nuisance-invariant-end-to-end-speech","title":"NIESR: Nuisance Invariant End-to-end Speech Recognition","date":"2019-07-07","arxiv_id":"1907.03233","repositories_listed":1,"syntology":null},{"url":"/paper/attention-model-for-articulatory-features","slug":"attention-model-for-articulatory-features","title":"Attention model for articulatory features detection","date":"2019-07-02","arxiv_id":"1907.01914","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-phonetic-pretraining-for-end-to","slug":"contextual-phonetic-pretraining-for-end-to","title":"BERTphone: Phonetically-Aware Encoder Representations for Utterance-Level Speaker and Language Recognition","date":"2019-06-30","arxiv_id":"1907.00457","repositories_listed":1,"syntology":null},{"url":"/paper/parzen-filters-for-spectral-decomposition-of","slug":"parzen-filters-for-spectral-decomposition-of","title":"Learning Waveform-Based Acoustic Models using Deep Variational Convolutional Neural Networks","date":"2019-06-23","arxiv_id":"1906.09526","repositories_listed":1,"syntology":null},{"url":"/paper/decoding-p300-variability-using-convolutional","slug":"decoding-p300-variability-using-convolutional","title":"Decoding P300 Variability using Convolutional Neural Networks","date":"2019-06-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/enabling-real-time-neural-ime-with","slug":"enabling-real-time-neural-ime-with","title":"Enabling Real-time Neural IME with Incremental Vocabulary Selection","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/guided-source-separation-meets-a-strong-asr","slug":"guided-source-separation-meets-a-strong-asr","title":"Guided Source Separation Meets a Strong ASR Backend: Hitachi/Paderborn University Joint Investigation for Dinner Party ASR","date":"2019-05-29","arxiv_id":"1905.12230","repositories_listed":1,"syntology":null},{"url":"/paper/learning-optimal-data-augmentation-policies","slug":"learning-optimal-data-augmentation-policies","title":"Learning Optimal Data Augmentation Policies via Bayesian Optimization for Image Classification Tasks","date":"2019-05-06","arxiv_id":"1905.02610","repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-for-audio-signal-processing","slug":"deep-learning-for-audio-signal-processing","title":"Deep Learning for Audio Signal Processing","date":"2019-04-30","arxiv_id":"1905.00078","repositories_listed":1,"syntology":null},{"url":"/paper/phonetically-oriented-word-error-alignment","slug":"phonetically-oriented-word-error-alignment","title":"Phonetically-Oriented Word Error Alignment for Speech Recognition Error Analysis in Speech Translation","date":"2019-04-24","arxiv_id":"1904.11024","repositories_listed":1,"syntology":null},{"url":"/paper/realizing-petabyte-scale-acoustic-modeling","slug":"realizing-petabyte-scale-acoustic-modeling","title":"Realizing Petabyte Scale Acoustic Modeling","date":"2019-04-24","arxiv_id":"1904.10584","repositories_listed":1,"syntology":null},{"url":"/paper/the-speechtransformer-for-large-scale","slug":"the-speechtransformer-for-large-scale","title":"The Speechtransformer for Large-scale Mandarin Chinese Speech Recognition","date":"2019-04-17","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/crf-based-single-stage-acoustic-modeling-with","slug":"crf-based-single-stage-acoustic-modeling-with","title":"CRF-based Single-stage Acoustic Modeling with CTC Topology","date":"2019-04-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-speech-domain-adaptation-based","slug":"unsupervised-speech-domain-adaptation-based","title":"Unsupervised Speech Domain Adaptation Based on Disentangled Representation Learning for Robust Speech Recognition","date":"2019-04-12","arxiv_id":"1904.06086","repositories_listed":1,"syntology":null},{"url":"/paper/a-target-agnostic-attack-on-deep-models","slug":"a-target-agnostic-attack-on-deep-models","title":"A Target-Agnostic Attack on Deep Models: Exploiting Security Vulnerabilities of Transfer Learning","date":"2019-04-08","arxiv_id":"1904.04334","repositories_listed":1,"syntology":null},{"url":"/paper/enriching-rare-word-representations-in-neural","slug":"enriching-rare-word-representations-in-neural","title":"Enriching Rare Word Representations in Neural Language Models by Embedding Matrix Augmentation","date":"2019-04-08","arxiv_id":"1904.03799","repositories_listed":1,"syntology":null},{"url":"/paper/spoken-language-intent-detection-using","slug":"spoken-language-intent-detection-using","title":"Spoken Language Intent Detection using Confusion2Vec","date":"2019-04-07","arxiv_id":"1904.03576","repositories_listed":1,"syntology":null},{"url":"/paper/imperceptible-robust-and-targeted-adversarial","slug":"imperceptible-robust-and-targeted-adversarial","title":"Imperceptible, Robust, and Targeted Adversarial Examples for Automatic Speech Recognition","date":"2019-03-22","arxiv_id":"1903.10346","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-sequence-to-sequence-models-for","slug":"evaluating-sequence-to-sequence-models-for","title":"Evaluating Sequence-to-Sequence Models for Handwritten Text Recognition","date":"2019-03-18","arxiv_id":"1903.07377","repositories_listed":1,"syntology":null},{"url":"/paper/audiovisual-speaker-tracking-using-nonlinear","slug":"audiovisual-speaker-tracking-using-nonlinear","title":"Audiovisual Speaker Tracking using Nonlinear Dynamical Systems with Dynamic Stream Weights","date":"2019-03-14","arxiv_id":"1903.06031","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-speech-recognition-using-a-high","slug":"end-to-end-speech-recognition-using-a-high","title":"End-To-End Speech Recognition Using A High Rank LSTM-CTC Based Model","date":"2019-03-12","arxiv_id":"1903.05261","repositories_listed":1,"syntology":null},{"url":"/paper/kt-speech-crawler-automatic-dataset","slug":"kt-speech-crawler-automatic-dataset","title":"KT-Speech-Crawler: Automatic Dataset Construction for Speech Recognition from YouTube Videos","date":"2019-03-01","arxiv_id":"1903.00216","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/kt-speech-crawler-automatic-dataset#ran","syntology_url":"https://syntology.ai/paper/1903.00216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1903.00216"}},"official":{"repos":["EgorLakomkin/KTSpeechCrawler"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stfnets-learning-sensing-signals-from-the","slug":"stfnets-learning-sensing-signals-from-the","title":"STFNets: Learning Sensing Signals from the Time-Frequency Perspective with Short-Time Fourier Neural Networks","date":"2019-02-21","arxiv_id":"1902.07849","repositories_listed":1,"syntology":null},{"url":"/paper/audio-linguistic-embeddings-for-spoken","slug":"audio-linguistic-embeddings-for-spoken","title":"Audio-Linguistic Embeddings for Spoken Sentences","date":"2019-02-20","arxiv_id":"1902.07817","repositories_listed":1,"syntology":null},{"url":"/paper/a-fully-differentiable-beam-search-decoder","slug":"a-fully-differentiable-beam-search-decoder","title":"A Fully Differentiable Beam Search Decoder","date":"2019-02-16","arxiv_id":"1902.06022","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/a-fully-differentiable-beam-search-decoder#ran","syntology_url":"https://syntology.ai/paper/1902.06022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.06022"}},"official":null}},{"url":"/paper/harnessing-gans-for-addition-of-new-classes","slug":"harnessing-gans-for-addition-of-new-classes","title":"Harnessing GANs for Zero-shot Learning of New Classes in Visual Speech Recognition","date":"2019-01-29","arxiv_id":"1901.10139","repositories_listed":1,"syntology":null},{"url":"/paper/self-attention-networks-for-connectionist","slug":"self-attention-networks-for-connectionist","title":"Self-Attention Networks for Connectionist Temporal Classification in Speech Recognition","date":"2019-01-22","arxiv_id":"1901.10055","repositories_listed":1,"syntology":null},{"url":"/paper/fastgrnn-a-fast-accurate-stable-and-tiny","slug":"fastgrnn-a-fast-accurate-stable-and-tiny","title":"FastGRNN: A Fast, Accurate, Stable and Tiny Kilobyte Sized Gated Recurrent Neural Network","date":"2019-01-08","arxiv_id":"1901.02358","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-spectro-temporal-features-in-end-to","slug":"exploring-spectro-temporal-features-in-end-to","title":"Exploring spectro-temporal features in end-to-end convolutional neural networks","date":"2019-01-01","arxiv_id":"1901.00072","repositories_listed":1,"syntology":null},{"url":"/paper/connectionist-temporal-classification-with","slug":"connectionist-temporal-classification-with","title":"Connectionist Temporal Classification with Maximum Entropy Regularization","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/context-aware-dialog-re-ranking-for-task","slug":"context-aware-dialog-re-ranking-for-task","title":"Context-Aware Dialog Re-Ranking for Task-Oriented Dialog Systems","date":"2018-11-28","arxiv_id":"1811.11430","repositories_listed":1,"syntology":null},{"url":"/paper/interpretable-convolutional-filters-with","slug":"interpretable-convolutional-filters-with","title":"Interpretable Convolutional Filters with SincNet","date":"2018-11-23","arxiv_id":"1811.09725","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-effects-of-word","slug":"investigating-the-effects-of-word","title":"Investigating the Effects of Word Substitution Errors on Sentence Embeddings","date":"2018-11-16","arxiv_id":"1811.07021","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-grounding-for-sequence-to-sequence","slug":"multimodal-grounding-for-sequence-to-sequence","title":"Multimodal Grounding for Sequence-to-Sequence Speech Recognition","date":"2018-11-09","arxiv_id":"1811.03865","repositories_listed":1,"syntology":null},{"url":"/paper/unpaired-speech-enhancement-by-acoustic-and","slug":"unpaired-speech-enhancement-by-acoustic-and","title":"Unpaired Speech Enhancement by Acoustic and Adversarial Supervision for Speech Recognition","date":"2018-11-06","arxiv_id":"1811.02182","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-end-to-end-solution-to-mandarin","slug":"on-the-end-to-end-solution-to-mandarin","title":"On the End-to-End Solution to Mandarin-English Code-switching Speech Recognition","date":"2018-11-01","arxiv_id":"1811.00241","repositories_listed":1,"syntology":null},{"url":"/paper/attention-based-sequence-to-sequence-model","slug":"attention-based-sequence-to-sequence-model","title":"Attention-based sequence-to-sequence model for speech recognition: development of state-of-the-art system on LibriSpeech and its application to non-native English","date":"2018-10-31","arxiv_id":"1810.13088","repositories_listed":1,"syntology":null},{"url":"/paper/language-modeling-for-code-switching","slug":"language-modeling-for-code-switching","title":"Language Modeling for Code-Switching: Evaluation, Integration of Monolingual Data, and Discriminative Training","date":"2018-10-28","arxiv_id":"1810.11895","repositories_listed":1,"syntology":null},{"url":"/paper/robust-audio-adversarial-example-for-a","slug":"robust-audio-adversarial-example-for-a","title":"Robust Audio Adversarial Example for a Physical Attack","date":"2018-10-28","arxiv_id":"1810.11793","repositories_listed":1,"syntology":null},{"url":"/paper/interpretable-convolutional-filters-with-1","slug":"interpretable-convolutional-filters-with-1","title":"Interpretable Convolutional Filters with SincNet","date":"2018-10-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-text-generation-using-an-outline","slug":"hierarchical-text-generation-using-an-outline","title":"Hierarchical Text Generation using an Outline","date":"2018-10-20","arxiv_id":"1810.08802","repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-stochastic-gradient-descent-for","slug":"evolutionary-stochastic-gradient-descent-for","title":"Evolutionary Stochastic Gradient Descent for Optimization of Deep Neural Networks","date":"2018-10-16","arxiv_id":"1810.06773","repositories_listed":1,"syntology":null},{"url":"/paper/extended-bit-plane-compression-for","slug":"extended-bit-plane-compression-for","title":"Extended Bit-Plane Compression for Convolutional Neural Network Accelerators","date":"2018-10-01","arxiv_id":"1810.03979","repositories_listed":1,"syntology":null},{"url":"/paper/multiwoz-a-large-scale-multi-domain-wizard-of-1","slug":"multiwoz-a-large-scale-multi-domain-wizard-of-1","title":"MultiWOZ - A Large-Scale Multi-Domain Wizard-of-Oz Dataset for Task-Oriented Dialogue Modelling","date":"2018-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/nice-noise-injection-and-clamping-estimation","slug":"nice-noise-injection-and-clamping-estimation","title":"NICE: Noise Injection and Clamping Estimation for Neural Network Quantization","date":"2018-09-29","arxiv_id":"1810.00162","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/nice-noise-injection-and-clamping-estimation#ran","syntology_url":"https://syntology.ai/paper/1810.00162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.00162"}},"official":{"repos":["Lancer555/NICE"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/low-frequency-adversarial-perturbation","slug":"low-frequency-adversarial-perturbation","title":"Low Frequency Adversarial Perturbation","date":"2018-09-24","arxiv_id":"1809.08758","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/low-frequency-adversarial-perturbation#ran","syntology_url":"https://syntology.ai/paper/1809.08758","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.08758"}},"official":{"repos":["cg563/low-frequency-adversarial"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-training-on-high-resource-speech","slug":"pre-training-on-high-resource-speech","title":"Pre-training on high-resource speech recognition improves low-resource speech-to-text translation","date":"2018-09-05","arxiv_id":"1809.01431","repositories_listed":1,"syntology":null},{"url":"/paper/lrs3-ted-a-large-scale-dataset-for-visual","slug":"lrs3-ted-a-large-scale-dataset-for-visual","title":"LRS3-TED: a large-scale dataset for visual speech recognition","date":"2018-09-03","arxiv_id":"1809.00496","repositories_listed":1,"syntology":null},{"url":"/paper/semi-orthogonal-low-rank-matrix-factorization","slug":"semi-orthogonal-low-rank-matrix-factorization","title":"Semi-Orthogonal Low-Rank Matrix Factorization for Deep Neural Networks","date":"2018-09-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-adapt-a-meta-learning-approach","slug":"learning-to-adapt-a-meta-learning-approach","title":"Learning to adapt: a meta-learning approach for speaker adaptation","date":"2018-08-30","arxiv_id":"1808.10239","repositories_listed":1,"syntology":null},{"url":"/paper/wisebe-window-based-sentence-boundary","slug":"wisebe-window-based-sentence-boundary","title":"WiSeBE: Window-based Sentence Boundary Evaluation","date":"2018-08-27","arxiv_id":"1808.08850","repositories_listed":1,"syntology":null},{"url":"/paper/neural-architecture-search-a-survey","slug":"neural-architecture-search-a-survey","title":"Neural Architecture Search: A Survey","date":"2018-08-16","arxiv_id":"1808.05377","repositories_listed":1,"syntology":null},{"url":"/paper/on-device-neural-language-model-based-word","slug":"on-device-neural-language-model-based-word","title":"On-Device Neural Language Model Based Word Prediction","date":"2018-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-comparison-of-techniques-for-language-model","slug":"a-comparison-of-techniques-for-language-model","title":"A Comparison of Techniques for Language Model Integration in Encoder-Decoder Speech Recognition","date":"2018-07-27","arxiv_id":"1807.10857","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-keyword-spotting-for-visual-speech","slug":"zero-shot-keyword-spotting-for-visual-speech","title":"Zero-shot keyword spotting for visual speech recognition in-the-wild","date":"2018-07-23","arxiv_id":"1807.08469","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparison-of-adaptation-techniques-and","slug":"a-comparison-of-adaptation-techniques-and","title":"A Comparison of Adaptation Techniques and Recurrent Neural Network Architectures","date":"2018-07-12","arxiv_id":"1807.06441","repositories_listed":1,"syntology":null},{"url":"/paper/improving-slot-filling-in-spoken-language","slug":"improving-slot-filling-in-spoken-language","title":"Improving Slot Filling in Spoken Language Understanding with Joint Pointer and Attention","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/word-error-rate-estimation-for-speech","slug":"word-error-rate-estimation-for-speech","title":"Word Error Rate Estimation for Speech Recognition: e-WER","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-gammatone-frequency-cepstral","slug":"evaluating-gammatone-frequency-cepstral","title":"Evaluating Gammatone Frequency Cepstral Coefficients with Neural Networks for Emotion Recognition from Speech","date":"2018-06-23","arxiv_id":"1806.09010","repositories_listed":1,"syntology":null},{"url":"/paper/quaternion-convolutional-neural-networks-for-1","slug":"quaternion-convolutional-neural-networks-for-1","title":"Quaternion Convolutional Neural Networks for End-to-End Automatic Speech Recognition","date":"2018-06-20","arxiv_id":"1806.07789","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-of-recent-dnn-architectures-on-the","slug":"a-survey-of-recent-dnn-architectures-on-the","title":"A Survey of Recent DNN Architectures on the TIMIT Phone Recognition Task","date":"2018-06-19","arxiv_id":"1806.07974","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-speech-recognition-from-the-raw","slug":"end-to-end-speech-recognition-from-the-raw","title":"End-to-End Speech Recognition From the Raw Waveform","date":"2018-06-19","arxiv_id":"1806.07098","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-dnns-and-its-ensembles-on-the-timit","slug":"recurrent-dnns-and-its-ensembles-on-the-timit","title":"Recurrent DNNs and its Ensembles on the TIMIT Phone Recognition Task","date":"2018-06-19","arxiv_id":"1806.07186","repositories_listed":1,"syntology":null},{"url":"/paper/lstm-benchmarks-for-deep-learning-frameworks","slug":"lstm-benchmarks-for-deep-learning-frameworks","title":"LSTM Benchmarks for Deep Learning Frameworks","date":"2018-06-05","arxiv_id":"1806.01818","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-macromicro-level-backpropagation-for","slug":"hybrid-macromicro-level-backpropagation-for","title":"Hybrid Macro/Micro Level Backpropagation for Training Deep Spiking Neural Networks","date":"2018-05-21","arxiv_id":"1805.07866","repositories_listed":1,"syntology":null},{"url":"/paper/targeted-adversarial-examples-for-black-box","slug":"targeted-adversarial-examples-for-black-box","title":"Targeted Adversarial Examples for Black Box Audio Systems","date":"2018-05-20","arxiv_id":"1805.07820","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/targeted-adversarial-examples-for-black-box#ran","syntology_url":"https://syntology.ai/paper/1805.07820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.07820"}},"official":{"repos":["rtaori/Black-Box-Audio"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluation-phonemic-transcription-of-low","slug":"evaluation-phonemic-transcription-of-low","title":"Evaluation Phonemic Transcription of Low-Resource Tonal Languages for Language Documentation","date":"2018-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/interpersonal-relationship-labels-for-the","slug":"interpersonal-relationship-labels-for-the","title":"Interpersonal Relationship Labels for the CALLHOME Corpus","date":"2018-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/preparing-data-from-psychotherapy-for-natural","slug":"preparing-data-from-psychotherapy-for-natural","title":"Preparing Data from Psychotherapy for Natural Language Processing","date":"2018-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/tf-lm-tensorflow-based-language-modeling","slug":"tf-lm-tensorflow-based-language-modeling","title":"TF-LM: TensorFlow-based Language Modeling Toolkit","date":"2018-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/syllable-based-sequence-to-sequence-speech","slug":"syllable-based-sequence-to-sequence-speech","title":"Syllable-Based Sequence-to-Sequence Speech Recognition with the Transformer in Mandarin Chinese","date":"2018-04-28","arxiv_id":"1804.10752","repositories_listed":1,"syntology":null},{"url":"/paper/optimus-an-efficient-dynamic-resource","slug":"optimus-an-efficient-dynamic-resource","title":"Optimus: An Efficient Dynamic Resource Scheduler for Deep Learning Clusters","date":"2018-04-26","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/attentive-sequence-to-sequence-learning-for","slug":"attentive-sequence-to-sequence-learning-for","title":"Attentive Sequence-to-Sequence Learning for Diacritic Restoration of Yorùbá Language Text","date":"2018-04-03","arxiv_id":"1804.00832","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-generative-adversarial-networks","slug":"investigating-generative-adversarial-networks","title":"Investigating Generative Adversarial Networks based Speech Dereverberation for Robust Speech Recognition","date":"2018-03-27","arxiv_id":"1803.10132","repositories_listed":1,"syntology":null},{"url":"/paper/light-gated-recurrent-units-for-speech","slug":"light-gated-recurrent-units-for-speech","title":"Light Gated Recurrent Units for Speech Recognition","date":"2018-03-26","arxiv_id":"1803.10225","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-bottleneck-features-for-subword","slug":"multilingual-bottleneck-features-for-subword","title":"Multilingual bottleneck features for subword modeling in zero-resource languages","date":"2018-03-23","arxiv_id":"1803.08863","repositories_listed":1,"syntology":null}],"record_sha256":"0a1bcd2ff9d8c135053eea467d5f17b00ee13adc3eef9cbe69856cd97efa918f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}