{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/2","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":58,"rows_per_page":100,"rows":[101,200],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1","next":"/task/speech-recognition-1/papers/3","papers":[{"url":"/paper/the-npu-aslp-liauto-system-description-for","slug":"the-npu-aslp-liauto-system-description-for","title":"The NPU-ASLP-LiAuto System Description for Visual Speech Recognition in CNVSRC 2023","date":"2024-01-07","arxiv_id":"2401.06788","repositories_listed":2,"syntology":null},{"url":"/paper/automatic-disfluency-detection-from","slug":"automatic-disfluency-detection-from","title":"Automatic Disfluency Detection from Untranscribed Speech","date":"2023-11-01","arxiv_id":"2311.00867","repositories_listed":2,"syntology":null},{"url":"/paper/distil-whisper-robust-knowledge-distillation","slug":"distil-whisper-robust-knowledge-distillation","title":"Distil-Whisper: Robust Knowledge Distillation via Large-Scale Pseudo Labelling","date":"2023-11-01","arxiv_id":"2311.00430","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/distil-whisper-robust-knowledge-distillation#ran","syntology_url":"https://syntology.ai/paper/2311.00430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.00430"}},"official":{"repos":["huggingface/distil-whisper"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/back-transcription-as-a-method-for-evaluating","slug":"back-transcription-as-a-method-for-evaluating","title":"Back Transcription as a Method for Evaluating Robustness of Natural Language Understanding Models to Speech Recognition Errors","date":"2023-10-25","arxiv_id":"2310.16609","repositories_listed":2,"syntology":null},{"url":"/paper/lauragpt-listen-attend-understand-and","slug":"lauragpt-listen-attend-understand-and","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","date":"2023-10-07","arxiv_id":"2310.04673","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lauragpt-listen-attend-understand-and#ran","syntology_url":"https://syntology.ai/paper/2310.04673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04673"}},"official":null}},{"url":"/paper/librispeech-pc-benchmark-for-evaluation-of","slug":"librispeech-pc-benchmark-for-evaluation-of","title":"LibriSpeech-PC: Benchmark for Evaluation of Punctuation and Capitalization Capabilities of end-to-end ASR Models","date":"2023-10-04","arxiv_id":"2310.02943","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/librispeech-pc-benchmark-for-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2310.02943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02943"}},"official":null}},{"url":"/paper/encodecmae-leveraging-neural-codecs-for","slug":"encodecmae-leveraging-neural-codecs-for","title":"EnCodecMAE: Leveraging neural codecs for universal audio representation learning","date":"2023-09-14","arxiv_id":"2309.07391","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/encodecmae-leveraging-neural-codecs-for#ran","syntology_url":"https://syntology.ai/paper/2309.07391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07391"}},"official":{"repos":["habla-liaa/encodecmae"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/promptasr-for-contextualized-asr-with","slug":"promptasr-for-contextualized-asr-with","title":"PromptASR for contextualized ASR with controllable style","date":"2023-09-14","arxiv_id":"2309.07414","repositories_listed":2,"syntology":null},{"url":"/paper/conformer-based-target-speaker-automatic","slug":"conformer-based-target-speaker-automatic","title":"Conformer-based Target-Speaker Automatic Speech Recognition for Single-Channel Audio","date":"2023-08-09","arxiv_id":"2308.05218","repositories_listed":2,"syntology":null},{"url":"/paper/learning-multi-modal-representations-by","slug":"learning-multi-modal-representations-by","title":"Learning Multi-modal Representations by Watching Hundreds of Surgical Video Lectures","date":"2023-07-27","arxiv_id":"2307.15220","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/learning-multi-modal-representations-by#ran","syntology_url":"https://syntology.ai/paper/2307.15220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.15220"}},"official":{"repos":["camma-public/peskavlp","camma-public/surgvlp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/ivrit-ai-a-comprehensive-dataset-of-hebrew","slug":"ivrit-ai-a-comprehensive-dataset-of-hebrew","title":"ivrit.ai: A Comprehensive Dataset of Hebrew Speech for AI Research and Development","date":"2023-07-17","arxiv_id":"2307.08720","repositories_listed":2,"syntology":null},{"url":"/paper/quilt-1m-one-million-image-text-pairs-for-1","slug":"quilt-1m-one-million-image-text-pairs-for-1","title":"Quilt-1M: One Million Image-Text Pairs for Histopathology","date":"2023-06-20","arxiv_id":"2306.11207","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quilt-1m-one-million-image-text-pairs-for-1#ran","syntology_url":"https://syntology.ai/paper/2306.11207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11207"}},"official":{"repos":["wisdomikezogwo/quilt1m"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unit-based-speech-to-speech-translation","slug":"unit-based-speech-to-speech-translation","title":"Textless Speech-to-Speech Translation With Limited Parallel Data","date":"2023-05-24","arxiv_id":"2305.15405","repositories_listed":2,"syntology":null},{"url":"/paper/exploring-energy-based-language-models-with","slug":"exploring-energy-based-language-models-with","title":"Exploring Energy-based Language Models with Different Architectures and Training Methods for Speech Recognition","date":"2023-05-22","arxiv_id":"2305.12676","repositories_listed":2,"syntology":null},{"url":"/paper/a-new-benchmark-of-aphasia-speech-recognition","slug":"a-new-benchmark-of-aphasia-speech-recognition","title":"A New Benchmark of Aphasia Speech Recognition and Detection Based on E-Branchformer and Multi-task Learning","date":"2023-05-19","arxiv_id":"2305.13331","repositories_listed":2,"syntology":null},{"url":"/paper/a-comparative-study-on-e-branchformer-vs","slug":"a-comparative-study-on-e-branchformer-vs","title":"A Comparative Study on E-Branchformer vs Conformer in Speech Recognition, Translation, and Understanding Tasks","date":"2023-05-18","arxiv_id":"2305.11073","repositories_listed":2,"syntology":null},{"url":"/paper/x-llm-bootstrapping-advanced-large-language","slug":"x-llm-bootstrapping-advanced-large-language","title":"X-LLM: Bootstrapping Advanced Large Language Models by Treating Multi-Modalities as Foreign Languages","date":"2023-05-07","arxiv_id":"2305.04160","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/x-llm-bootstrapping-advanced-large-language#ran","syntology_url":"https://syntology.ai/paper/2305.04160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04160"}},"official":null}},{"url":"/paper/reproducibility-is-nothing-without","slug":"reproducibility-is-nothing-without","title":"When Good and Reproducible Results are a Giant with Feet of Clay: The Importance of Software Quality in NLP","date":"2023-03-28","arxiv_id":"2303.16166","repositories_listed":2,"syntology":null},{"url":"/paper/auto-avsr-audio-visual-speech-recognition","slug":"auto-avsr-audio-visual-speech-recognition","title":"Auto-AVSR: Audio-Visual Speech Recognition with Automatic Labels","date":"2023-03-25","arxiv_id":"2303.14307","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/auto-avsr-audio-visual-speech-recognition#ran","syntology_url":"https://syntology.ai/paper/2303.14307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14307"}},"official":{"repos":["mpc001/auto_avsr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-transfer-from-pre-trained-language","slug":"knowledge-transfer-from-pre-trained-language","title":"Knowledge Transfer from Pre-trained Language Models to Cif-based Speech Recognizers via Hierarchical Distillation","date":"2023-01-30","arxiv_id":"2301.13003","repositories_listed":2,"syntology":null},{"url":"/paper/unsupervised-data-selection-for-tts-using","slug":"unsupervised-data-selection-for-tts-using","title":"Unsupervised Data Selection for TTS: Using Arabic Broadcast News as a Case Study","date":"2023-01-22","arxiv_id":"2301.09099","repositories_listed":2,"syntology":null},{"url":"/paper/a-persian-asr-based-ser-modification-of","slug":"a-persian-asr-based-ser-modification-of","title":"A Persian ASR-based SER: Modification of Sharif Emotional Speech Database and Investigation of Persian Text Corpora","date":"2022-11-18","arxiv_id":"2211.09956","repositories_listed":2,"syntology":null},{"url":"/paper/speechut-bridging-speech-and-text-with-hidden","slug":"speechut-bridging-speech-and-text-with-hidden","title":"SpeechUT: Bridging Speech and Text with Hidden-Unit for Encoder-Decoder Based Speech-Text Pre-training","date":"2022-10-07","arxiv_id":"2210.03730","repositories_listed":2,"syntology":null},{"url":"/paper/cmgan-conformer-based-metric-gan-for-monaural","slug":"cmgan-conformer-based-metric-gan-for-monaural","title":"CMGAN: Conformer-Based Metric-GAN for Monaural Speech Enhancement","date":"2022-09-22","arxiv_id":"2209.11112","repositories_listed":2,"syntology":null},{"url":"/paper/improving-mandarin-speech-recogntion-with","slug":"improving-mandarin-speech-recogntion-with","title":"Improving Mandarin Speech Recogntion with Block-augmented Transformer","date":"2022-07-24","arxiv_id":"2207.11697","repositories_listed":2,"syntology":null},{"url":"/paper/tevr-improving-speech-recognition-by-token","slug":"tevr-improving-speech-recognition-by-token","title":"TEVR: Improving Speech Recognition by Token Entropy Variance Reduction","date":"2022-06-25","arxiv_id":"2206.12693","repositories_listed":2,"syntology":null},{"url":"/paper/paraformer-fast-and-accurate-parallel","slug":"paraformer-fast-and-accurate-parallel","title":"Paraformer: Fast and Accurate Parallel Transformer for Non-autoregressive End-to-End Speech Recognition","date":"2022-06-16","arxiv_id":"2206.08317","repositories_listed":2,"syntology":null},{"url":"/paper/soundspaces-2-0-a-simulation-platform-for","slug":"soundspaces-2-0-a-simulation-platform-for","title":"SoundSpaces 2.0: A Simulation Platform for Visual-Acoustic Learning","date":"2022-06-16","arxiv_id":"2206.08312","repositories_listed":2,"syntology":null},{"url":"/paper/towards-understanding-and-mitigating-audio","slug":"towards-understanding-and-mitigating-audio","title":"Towards Understanding and Mitigating Audio Adversarial Examples for Speaker Recognition","date":"2022-06-07","arxiv_id":"2206.03393","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-understanding-and-mitigating-audio#ran","syntology_url":"https://syntology.ai/paper/2206.03393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.03393"}},"official":null}},{"url":"/paper/how-does-pre-trained-wav2vec2-0-perform-on","slug":"how-does-pre-trained-wav2vec2-0-perform-on","title":"How Does Pre-trained Wav2Vec 2.0 Perform on Domain Shifted ASR? An Extensive Benchmark on Air Traffic Control Communications","date":"2022-03-31","arxiv_id":"2203.16822","repositories_listed":2,"syntology":null},{"url":"/paper/recent-improvements-of-asr-models-in-the-face","slug":"recent-improvements-of-asr-models-in-the-face","title":"Recent improvements of ASR models in the face of adversarial attacks","date":"2022-03-29","arxiv_id":"2203.16536","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recent-improvements-of-asr-models-in-the-face#ran","syntology_url":"https://syntology.ai/paper/2203.16536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16536"}},"official":{"repos":["raphaelolivier/robust_speech"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/listen-adapt-better-wer-source-free-single","slug":"listen-adapt-better-wer-source-free-single","title":"Listen, Adapt, Better WER: Source-free Single-utterance Test-time Adaptation for Automatic Speech Recognition","date":"2022-03-27","arxiv_id":"2203.14222","repositories_listed":2,"syntology":null},{"url":"/paper/a-3-t-alignment-aware-acoustic-and-text","slug":"a-3-t-alignment-aware-acoustic-and-text","title":"A$^3$T: Alignment-Aware Acoustic and Text Pretraining for Speech Synthesis and Editing","date":"2022-03-18","arxiv_id":"2203.09690","repositories_listed":2,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/a-3-t-alignment-aware-acoustic-and-text#ran","syntology_url":"https://syntology.ai/paper/2203.09690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09690"}},"official":null}},{"url":"/paper/visual-speech-recognition-for-multiple","slug":"visual-speech-recognition-for-multiple","title":"Visual Speech Recognition for Multiple Languages in the Wild","date":"2022-02-26","arxiv_id":"2202.13084","repositories_listed":2,"syntology":null},{"url":"/paper/learning-audio-visual-speech-representation-1","slug":"learning-audio-visual-speech-representation-1","title":"Learning Audio-Visual Speech Representation by Masked Multimodal Cluster Prediction","date":"2022-01-05","arxiv_id":"2201.02184","repositories_listed":2,"syntology":null},{"url":"/paper/xls-r-self-supervised-cross-lingual-speech","slug":"xls-r-self-supervised-cross-lingual-speech","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","date":"2021-11-17","arxiv_id":"2111.09296","repositories_listed":2,"syntology":null},{"url":"/paper/attention-based-multi-hypothesis-fusion-for","slug":"attention-based-multi-hypothesis-fusion-for","title":"Attention-based Multi-hypothesis Fusion for Speech Summarization","date":"2021-11-16","arxiv_id":"2111.08201","repositories_listed":2,"syntology":null},{"url":"/paper/coraa-a-large-corpus-of-spontaneous-and","slug":"coraa-a-large-corpus-of-spontaneous-and","title":"CORAA: a large corpus of spontaneous and prepared speech manually validated for speech recognition in Brazilian Portuguese","date":"2021-10-14","arxiv_id":"2110.15731","repositories_listed":2,"syntology":null},{"url":"/paper/bertraffic-a-robust-bert-based-approach-for","slug":"bertraffic-a-robust-bert-based-approach-for","title":"BERTraffic: BERT-based Joint Speaker Role and Speaker Change Detection for Air Traffic Control Communications","date":"2021-10-12","arxiv_id":"2110.05781","repositories_listed":2,"syntology":null},{"url":"/paper/interactive-feature-fusion-for-end-to-end","slug":"interactive-feature-fusion-for-end-to-end","title":"Interactive Feature Fusion for End-to-End Noise-Robust Speech Recognition","date":"2021-10-11","arxiv_id":"2110.05267","repositories_listed":2,"syntology":null},{"url":"/paper/k-wav2vec-2-0-automatic-speech-recognition","slug":"k-wav2vec-2-0-automatic-speech-recognition","title":"K-Wav2vec 2.0: Automatic Speech Recognition based on Joint Decoding of Graphemes and Syllables","date":"2021-10-11","arxiv_id":"2110.05172","repositories_listed":2,"syntology":null},{"url":"/paper/wenetspeech-a-10000-hours-multi-domain","slug":"wenetspeech-a-10000-hours-multi-domain","title":"WenetSpeech: A 10000+ Hours Multi-domain Mandarin Corpus for Speech Recognition","date":"2021-10-07","arxiv_id":"2110.03370","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wenetspeech-a-10000-hours-multi-domain#ran","syntology_url":"https://syntology.ai/paper/2110.03370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.03370"}},"official":{"repos":["wenet-e2e/wenetspeech"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/simple-and-effective-zero-shot-cross-lingual","slug":"simple-and-effective-zero-shot-cross-lingual","title":"Simple and Effective Zero-shot Cross-lingual Phoneme Recognition","date":"2021-09-23","arxiv_id":"2109.11680","repositories_listed":2,"syntology":null},{"url":"/paper/sveva-fair-a-framework-for-evaluating","slug":"sveva-fair-a-framework-for-evaluating","title":"SVEva Fair: A Framework for Evaluating Fairness in Speaker Verification","date":"2021-07-26","arxiv_id":"2107.12049","repositories_listed":2,"syntology":null},{"url":"/paper/a-comparison-of-methods-for-oov-word","slug":"a-comparison-of-methods-for-oov-word","title":"A Comparison of Methods for OOV-word Recognition on a New Public Dataset","date":"2021-07-16","arxiv_id":"2107.08091","repositories_listed":2,"syntology":null},{"url":"/paper/clsril-23-cross-lingual-speech","slug":"clsril-23-cross-lingual-speech","title":"CLSRIL-23: Cross Lingual Speech Representations for Indic Languages","date":"2021-07-15","arxiv_id":"2107.07402","repositories_listed":2,"syntology":null},{"url":"/paper/meshrir-a-dataset-of-room-impulse-responses","slug":"meshrir-a-dataset-of-room-impulse-responses","title":"MeshRIR: A Dataset of Room Impulse Responses on Meshed Grid Points For Evaluating Sound Field Analysis and Synthesis Methods","date":"2021-06-21","arxiv_id":"2106.10801","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meshrir-a-dataset-of-room-impulse-responses#ran","syntology_url":"https://syntology.ai/paper/2106.10801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.10801"}},"official":{"repos":["sh01k/MeshRIR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lightweight-adapter-tuning-for-multilingual","slug":"lightweight-adapter-tuning-for-multilingual","title":"Lightweight Adapter Tuning for Multilingual Speech Translation","date":"2021-06-02","arxiv_id":"2106.01463","repositories_listed":2,"syntology":null},{"url":"/paper/fedscale-benchmarking-model-and-system","slug":"fedscale-benchmarking-model-and-system","title":"FedScale: Benchmarking Model and System Performance of Federated Learning at Scale","date":"2021-05-24","arxiv_id":"2105.11367","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fedscale-benchmarking-model-and-system#ran","syntology_url":"https://syntology.ai/paper/2105.11367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.11367"}},"official":{"repos":["SymbioticLab/FedScale"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/exploiting-adapters-for-cross-lingual-low","slug":"exploiting-adapters-for-cross-lingual-low","title":"Exploiting Adapters for Cross-lingual Low-resource Speech Recognition","date":"2021-05-18","arxiv_id":"2105.11905","repositories_listed":2,"syntology":null},{"url":"/paper/emotion-recognition-from-speech-using-wav2vec","slug":"emotion-recognition-from-speech-using-wav2vec","title":"Emotion Recognition from Speech Using Wav2vec 2.0 Embeddings","date":"2021-04-08","arxiv_id":"2104.03502","repositories_listed":2,"syntology":null},{"url":"/paper/speechocean762-an-open-source-non-native","slug":"speechocean762-an-open-source-non-native","title":"speechocean762: An Open-Source Non-native English Speech Corpus For Pronunciation Assessment","date":"2021-04-03","arxiv_id":"2104.01378","repositories_listed":2,"syntology":null},{"url":"/paper/fdlp-spectrogram-capturing-speech-dynamics-in","slug":"fdlp-spectrogram-capturing-speech-dynamics-in","title":"Radically Old Way of Computing Spectra: Applications in End-to-End ASR","date":"2021-03-25","arxiv_id":"2103.14129","repositories_listed":2,"syntology":null},{"url":"/paper/domain-generalization-a-survey","slug":"domain-generalization-a-survey","title":"Domain Generalization: A Survey","date":"2021-03-03","arxiv_id":"2103.02503","repositories_listed":2,"syntology":null},{"url":"/paper/bembaspeech-a-speech-recognition-corpus-for","slug":"bembaspeech-a-speech-recognition-corpus-for","title":"BembaSpeech: A Speech Recognition Corpus for the Bemba Language","date":"2021-02-09","arxiv_id":"2102.04889","repositories_listed":2,"syntology":null},{"url":"/paper/understanding-the-tradeoffs-in-client-side","slug":"understanding-the-tradeoffs-in-client-side","title":"Understanding the Tradeoffs in Client-side Privacy for Downstream Speech Tasks","date":"2021-01-22","arxiv_id":"2101.08919","repositories_listed":2,"syntology":null},{"url":"/paper/kaleidoscope-an-efficient-learnable-1","slug":"kaleidoscope-an-efficient-learnable-1","title":"Kaleidoscope: An Efficient, Learnable Representation For All Structured Linear Maps","date":"2020-12-29","arxiv_id":"2012.14966","repositories_listed":2,"syntology":{"n":23,"n_ran":16,"n_constructed":9,"n_ran_checked":9,"n_instrument":7,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"16 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 7 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/kaleidoscope-an-efficient-learnable-1#ran","syntology_url":"https://syntology.ai/paper/2012.14966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.14966"}},"official":{"repos":["HazyResearch/butterfly","HazyResearch/learning-circuits"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":9,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/a-comprehensive-evaluation-of-incremental","slug":"a-comprehensive-evaluation-of-incremental","title":"A Comprehensive Evaluation of Incremental Speech Recognition and Diarization for Conversational AI","date":"2020-12-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/decentralizing-feature-extraction-with","slug":"decentralizing-feature-extraction-with","title":"Decentralizing Feature Extraction with Quantum Convolutional Neural Network for Automatic Speech Recognition","date":"2020-10-26","arxiv_id":"2010.13309","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/decentralizing-feature-extraction-with#ran","syntology_url":"https://syntology.ai/paper/2010.13309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.13309"}},"official":{"repos":["huckiyang/speech_quantum_dl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/towards-resistant-audio-adversarial-examples","slug":"towards-resistant-audio-adversarial-examples","title":"Towards Resistant Audio Adversarial Examples","date":"2020-10-14","arxiv_id":"2010.07190","repositories_listed":2,"syntology":null},{"url":"/paper/representation-learning-for-sequence-data-1","slug":"representation-learning-for-sequence-data-1","title":"Representation Learning for Sequence Data with Deep Autoencoding Predictive Components","date":"2020-10-07","arxiv_id":"2010.03135","repositories_listed":2,"syntology":{"n":29,"n_ran":22,"n_constructed":0,"n_ran_checked":22,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":22,"n_pointer_only":2,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 22 with no instrument failure: 0 honoured, 0 violated, 22 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/representation-learning-for-sequence-data-1#ran","syntology_url":"https://syntology.ai/paper/2010.03135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.03135"}},"official":{"repos":["JunwenBai/DAPC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed"]}}},{"url":"/paper/end-to-end-speech-recognition-and-disfluency","slug":"end-to-end-speech-recognition-and-disfluency","title":"End-to-End Speech Recognition and Disfluency Removal","date":"2020-09-22","arxiv_id":"2009.10298","repositories_listed":2,"syntology":null},{"url":"/paper/voicefilter-lite-streaming-targeted-voice","slug":"voicefilter-lite-streaming-targeted-voice","title":"VoiceFilter-Lite: Streaming Targeted Voice Separation for On-Device Speech Recognition","date":"2020-09-09","arxiv_id":"2009.04323","repositories_listed":2,"syntology":null},{"url":"/paper/computer-generated-music-for-tabletop-role","slug":"computer-generated-music-for-tabletop-role","title":"Computer-Generated Music for Tabletop Role-Playing Games","date":"2020-08-16","arxiv_id":"2008.07009","repositories_listed":2,"syntology":null},{"url":"/paper/pretraining-techniques-for-sequence-to","slug":"pretraining-techniques-for-sequence-to","title":"Pretraining Techniques for Sequence-to-Sequence Voice Conversion","date":"2020-08-07","arxiv_id":"2008.03088","repositories_listed":2,"syntology":null},{"url":"/paper/covost-2-a-massively-multilingual-speech-to","slug":"covost-2-a-massively-multilingual-speech-to","title":"CoVoST 2 and Massively Multilingual Speech-to-Text Translation","date":"2020-07-20","arxiv_id":"2007.10310","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/covost-2-a-massively-multilingual-speech-to#ran","syntology_url":"https://syntology.ai/paper/2007.10310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.10310"}},"official":{"repos":["facebookresearch/covost"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-lyrics-transcription-using-dilated","slug":"automatic-lyrics-transcription-using-dilated","title":"Automatic Lyrics Transcription using Dilated Convolutional Neural Networks with Self-Attention","date":"2020-07-13","arxiv_id":"2007.06486","repositories_listed":2,"syntology":null},{"url":"/paper/emotion-recognition-in-audio-and-video-using","slug":"emotion-recognition-in-audio-and-video-using","title":"Emotion Recognition in Audio and Video Using Deep Neural Networks","date":"2020-06-15","arxiv_id":"2006.08129","repositories_listed":2,"syntology":null},{"url":"/paper/an-adversarial-approach-for-explaining-the","slug":"an-adversarial-approach-for-explaining-the","title":"An Adversarial Approach for Explaining the Predictions of Deep Neural Networks","date":"2020-05-20","arxiv_id":"2005.10284","repositories_listed":2,"syntology":null},{"url":"/paper/discriminative-multi-modality-speech","slug":"discriminative-multi-modality-speech","title":"Discriminative Multi-modality Speech Recognition","date":"2020-05-12","arxiv_id":"2005.05592","repositories_listed":2,"syntology":null},{"url":"/paper/multilingual-twitter-corpus-and-baselines-for","slug":"multilingual-twitter-corpus-and-baselines-for","title":"Multilingual Twitter Corpus and Baselines for Evaluating Demographic Bias in Hate Speech Recognition","date":"2020-02-24","arxiv_id":"2002.10361","repositories_listed":2,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/multilingual-twitter-corpus-and-baselines-for#ran","syntology_url":"https://syntology.ai/paper/2002.10361","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.10361"}},"official":{"repos":["xiaoleihuang/Multilingual_Fairness_LREC"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/libri-light-a-benchmark-for-asr-with-limited","slug":"libri-light-a-benchmark-for-asr-with-limited","title":"Libri-Light: A Benchmark for ASR with Limited or No Supervision","date":"2019-12-17","arxiv_id":"1912.07875","repositories_listed":2,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/libri-light-a-benchmark-for-asr-with-limited#ran","syntology_url":"https://syntology.ai/paper/1912.07875","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.07875"}},"official":{"repos":["facebookresearch/libri-light"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/common-voice-a-massively-multilingual-speech","slug":"common-voice-a-massively-multilingual-speech","title":"Common Voice: A Massively-Multilingual Speech Corpus","date":"2019-12-13","arxiv_id":"1912.06670","repositories_listed":2,"syntology":null},{"url":"/paper/cat-crf-based-asr-toolkit","slug":"cat-crf-based-asr-toolkit","title":"CAT: CRF-based ASR Toolkit","date":"2019-11-20","arxiv_id":"1911.08747","repositories_listed":2,"syntology":null},{"url":"/paper/effectiveness-of-self-supervised-pre-training","slug":"effectiveness-of-self-supervised-pre-training","title":"Effectiveness of self-supervised pre-training for speech recognition","date":"2019-11-10","arxiv_id":"1911.03912","repositories_listed":2,"syntology":null},{"url":"/paper/confidence-estimation-for-black-box-automatic","slug":"confidence-estimation-for-black-box-automatic","title":"Confidence Estimation for Black Box Automatic Speech Recognition Systems Using Lattice Recurrent Neural Networks","date":"2019-10-25","arxiv_id":"1910.11933","repositories_listed":2,"syntology":null},{"url":"/paper/generative-pre-training-for-speech-with","slug":"generative-pre-training-for-speech-with","title":"Generative Pre-Training for Speech with Autoregressive Predictive Coding","date":"2019-10-23","arxiv_id":"1910.12607","repositories_listed":2,"syntology":null},{"url":"/paper/speech-vgg-a-deep-feature-extractor-for","slug":"speech-vgg-a-deep-feature-extractor-for","title":"Word-level Embeddings for Cross-Task Transfer Learning in Speech Processing","date":"2019-10-22","arxiv_id":"1910.09909","repositories_listed":2,"syntology":null},{"url":"/paper/dover-a-method-for-combining-diarization","slug":"dover-a-method-for-combining-diarization","title":"DOVER: A Method for Combining Diarization Outputs","date":"2019-09-17","arxiv_id":"1909.08090","repositories_listed":2,"syntology":null},{"url":"/paper/a-comparative-study-on-transformer-vs-rnn-in","slug":"a-comparative-study-on-transformer-vs-rnn-in","title":"A Comparative Study on Transformer vs RNN in Speech Applications","date":"2019-09-13","arxiv_id":"1909.06317","repositories_listed":2,"syntology":null},{"url":"/paper/ebpc-extended-bit-plane-compression-for-deep","slug":"ebpc-extended-bit-plane-compression-for-deep","title":"EBPC: Extended Bit-Plane Compression for Deep Neural Network Inference and Training Accelerators","date":"2019-08-30","arxiv_id":"1908.11645","repositories_listed":2,"syntology":null},{"url":"/paper/personal-vad-speaker-conditioned-voice","slug":"personal-vad-speaker-conditioned-voice","title":"Personal VAD: Speaker-Conditioned Voice Activity Detection","date":"2019-08-12","arxiv_id":"1908.04284","repositories_listed":2,"syntology":null},{"url":"/paper/delta-a-deep-learning-based-language","slug":"delta-a-deep-learning-based-language","title":"DELTA: A DEep learning based Language Technology plAtform","date":"2019-08-02","arxiv_id":"1908.01853","repositories_listed":2,"syntology":null},{"url":"/paper/cif-continuous-integrate-and-fire-for-end-to","slug":"cif-continuous-integrate-and-fire-for-end-to","title":"CIF: Continuous Integrate-and-Fire for End-to-End Speech Recognition","date":"2019-05-27","arxiv_id":"1905.11235","repositories_listed":2,"syntology":null},{"url":"/paper/rwth-asr-systems-for-librispeech-hybrid-vs","slug":"rwth-asr-systems-for-librispeech-hybrid-vs","title":"RWTH ASR Systems for LibriSpeech: Hybrid vs Attention -- w/o Data Augmentation","date":"2019-05-08","arxiv_id":"1905.03072","repositories_listed":2,"syntology":null},{"url":"/paper/mitigating-the-impact-of-speech-recognition-1","slug":"mitigating-the-impact-of-speech-recognition-1","title":"Mitigating the Impact of Speech Recognition Errors on Spoken Question Answering by Adversarial Domain Adaptation","date":"2019-04-16","arxiv_id":"1904.07904","repositories_listed":2,"syntology":null},{"url":"/paper/speech-and-speaker-recognition-from-raw","slug":"speech-and-speaker-recognition-from-raw","title":"Speech and Speaker Recognition from Raw Waveform with SincNet","date":"2018-12-13","arxiv_id":"1812.05920","repositories_listed":2,"syntology":null},{"url":"/paper/streaming-end-to-end-speech-recognition-for","slug":"streaming-end-to-end-speech-recognition-for","title":"Streaming End-to-end Speech Recognition For Mobile Devices","date":"2018-11-15","arxiv_id":"1811.06621","repositories_listed":2,"syntology":null},{"url":"/paper/bidirectional-quaternion-long-short-term","slug":"bidirectional-quaternion-long-short-term","title":"Bidirectional Quaternion Long-Short Term Memory Recurrent Neural Networks for Speech Recognition","date":"2018-11-06","arxiv_id":"1811.02566","repositories_listed":2,"syntology":null},{"url":"/paper/how2-a-large-scale-dataset-for-multimodal","slug":"how2-a-large-scale-dataset-for-multimodal","title":"How2: A Large-scale Dataset for Multimodal Language Understanding","date":"2018-11-01","arxiv_id":"1811.00347","repositories_listed":2,"syntology":null},{"url":"/paper/lrw-1000-a-naturally-distributed-large-scale","slug":"lrw-1000-a-naturally-distributed-large-scale","title":"LRW-1000: A Naturally-Distributed Large-Scale Benchmark for Lip Reading in the Wild","date":"2018-10-16","arxiv_id":"1810.06990","repositories_listed":2,"syntology":null},{"url":"/paper/optimal-completion-distillation-for-sequence","slug":"optimal-completion-distillation-for-sequence","title":"Optimal Completion Distillation for Sequence Learning","date":"2018-10-02","arxiv_id":"1810.01398","repositories_listed":2,"syntology":null},{"url":"/paper/open-source-automatic-speech-recognition-for","slug":"open-source-automatic-speech-recognition-for","title":"Open Source Automatic Speech Recognition for German","date":"2018-07-26","arxiv_id":"1807.10311","repositories_listed":2,"syntology":null},{"url":"/paper/augmented-cyclic-adversarial-learning-for-low","slug":"augmented-cyclic-adversarial-learning-for-low","title":"Augmented Cyclic Adversarial Learning for Low Resource Domain Adaptation","date":"2018-07-01","arxiv_id":"1807.00374","repositories_listed":2,"syntology":null},{"url":"/paper/twin-regularization-for-online-speech","slug":"twin-regularization-for-online-speech","title":"Twin Regularization for online speech recognition","date":"2018-04-15","arxiv_id":"1804.05374","repositories_listed":2,"syntology":null},{"url":"/paper/scalable-factorized-hierarchical-variational","slug":"scalable-factorized-hierarchical-variational","title":"Scalable Factorized Hierarchical Variational Autoencoder Training","date":"2018-04-09","arxiv_id":"1804.03201","repositories_listed":2,"syntology":null},{"url":"/paper/end-to-end-audiovisual-speech-recognition","slug":"end-to-end-audiovisual-speech-recognition","title":"End-to-end Audiovisual Speech Recognition","date":"2018-02-18","arxiv_id":"1802.06424","repositories_listed":2,"syntology":null},{"url":"/paper/training-rnns-as-fast-as-cnns","slug":"training-rnns-as-fast-as-cnns","title":"Training RNNs as Fast as CNNs","date":"2018-01-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/letter-based-speech-recognition-with-gated","slug":"letter-based-speech-recognition-with-gated","title":"Letter-Based Speech Recognition with Gated ConvNets","date":"2017-12-22","arxiv_id":"1712.09444","repositories_listed":2,"syntology":null},{"url":"/paper/minimum-word-error-rate-training-for","slug":"minimum-word-error-rate-training-for","title":"Minimum Word Error Rate Training for Attention-based Sequence-to-Sequence Models","date":"2017-12-05","arxiv_id":"1712.01818","repositories_listed":2,"syntology":null}],"record_sha256":"9ebf2b12dd307adcf89536a72531c38d8ded9d0d5bf5e0fe554ca45bba90c7b1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}