{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition-2/papers/5","list_of":"/task/automatic-speech-recognition-2","task":"Automatic Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":32,"rows_per_page":100,"rows":[401,500],"of":3174,"counts":{"archive_papers_tagged":3174,"with_a_code_link":677,"where_syntology_ran_a_sample":79,"not_listed_spam_title":0,"listed":3174,"listed_where_code_ran":79,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":62,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":62,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition-2","prev":"/task/automatic-speech-recognition-2/papers/4","next":"/task/automatic-speech-recognition-2/papers/6","papers":[{"url":"/paper/watch-what-you-pretrain-for-targeted","slug":"watch-what-you-pretrain-for-targeted","title":"Watch What You Pretrain For: Targeted, Transferable Adversarial Examples on Self-Supervised Speech Recognition models","date":"2022-09-17","arxiv_id":"2209.13523","repositories_listed":1,"syntology":null},{"url":"/paper/an-automatic-speech-recognition-system-for","slug":"an-automatic-speech-recognition-system-for","title":"An Automatic Speech Recognition System for Bengali Language based on Wav2Vec2 and Transfer Learning","date":"2022-09-16","arxiv_id":"2209.08119","repositories_listed":1,"syntology":null},{"url":"/paper/non-autoregressive-error-correction-for-ctc","slug":"non-autoregressive-error-correction-for-ctc","title":"Non-autoregressive Error Correction for CTC-based ASR with Phone-conditioned Masked LM","date":"2022-09-08","arxiv_id":"2209.04062","repositories_listed":1,"syntology":null},{"url":"/paper/mlphon-a-multifunctional-grapheme-phoneme","slug":"mlphon-a-multifunctional-grapheme-phoneme","title":"Mlphon: A Multifunctional Grapheme-Phoneme Conversion Tool Using Finite State Transducers","date":"2022-09-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/semantically-meaningful-metrics-for-norwegian","slug":"semantically-meaningful-metrics-for-norwegian","title":"Semantically Meaningful Metrics for Norwegian ASR Systems","date":"2022-09-03","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-sparse-conformer-for-speech-recognition","slug":"deep-sparse-conformer-for-speech-recognition","title":"Deep Sparse Conformer for Speech Recognition","date":"2022-09-01","arxiv_id":"2209.00260","repositories_listed":1,"syntology":null},{"url":"/paper/indicsuperb-a-speech-processing-universal","slug":"indicsuperb-a-speech-processing-universal","title":"IndicSUPERB: A Speech Processing Universal Performance Benchmark for Indian languages","date":"2022-08-24","arxiv_id":"2208.11761","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/indicsuperb-a-speech-processing-universal#ran","syntology_url":"https://syntology.ai/paper/2208.11761","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.11761"}},"official":{"repos":["AI4Bharat/indicSUPERB"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-robustness-of-end-to-end-neural","slug":"analyzing-robustness-of-end-to-end-neural","title":"Analyzing Robustness of End-to-End Neural Models for Automatic Speech Recognition","date":"2022-08-17","arxiv_id":"2208.08509","repositories_listed":1,"syntology":null},{"url":"/paper/comparison-and-analysis-of-new-curriculum","slug":"comparison-and-analysis-of-new-curriculum","title":"Comparison and Analysis of New Curriculum Criteria for End-to-End ASR","date":"2022-08-10","arxiv_id":"2208.05782","repositories_listed":1,"syntology":null},{"url":"/paper/asr-error-correction-with-constrained","slug":"asr-error-correction-with-constrained","title":"ASR Error Correction with Constrained Decoding on Operation Prediction","date":"2022-08-09","arxiv_id":"2208.04641","repositories_listed":1,"syntology":null},{"url":"/paper/thai-wav2vec2-0-with-commonvoice-v8","slug":"thai-wav2vec2-0-with-commonvoice-v8","title":"Thai Wav2Vec2.0 with CommonVoice V8","date":"2022-08-09","arxiv_id":"2208.04799","repositories_listed":1,"syntology":null},{"url":"/paper/dent-ddsp-data-efficient-noisy-speech","slug":"dent-ddsp-data-efficient-noisy-speech","title":"DENT-DDSP: Data-efficient noisy speech generator using differentiable digital signal processors for explicit distortion modelling and noise-robust speech recognition","date":"2022-08-01","arxiv_id":"2208.00987","repositories_listed":1,"syntology":null},{"url":"/paper/domain-specific-wav2vec-2-0-fine-tuning-for","slug":"domain-specific-wav2vec-2-0-fine-tuning-for","title":"Domain Specific Wav2vec 2.0 Fine-tuning For The SE&R 2022 Challenge","date":"2022-07-29","arxiv_id":"2207.14418","repositories_listed":1,"syntology":null},{"url":"/paper/towards-transfer-learning-of-wav2vec-2-0-for","slug":"towards-transfer-learning-of-wav2vec-2-0-for","title":"Transfer Learning of wav2vec 2.0 for Automatic Lyric Transcription","date":"2022-07-20","arxiv_id":"2207.09747","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-transfer-learning-of-wav2vec-2-0-for#ran","syntology_url":"https://syntology.ai/paper/2207.09747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.09747"}},"official":{"repos":["guxm2021/alt_speechbrain"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/when-is-tts-augmentation-through-a-pivot","slug":"when-is-tts-augmentation-through-a-pivot","title":"When Is TTS Augmentation Through a Pivot Language Useful?","date":"2022-07-20","arxiv_id":"2207.09889","repositories_listed":1,"syntology":null},{"url":"/paper/espnet-se-speech-enhancement-for-robust","slug":"espnet-se-speech-enhancement-for-robust","title":"ESPnet-SE++: Speech Enhancement for Robust Speech Recognition, Translation, and Understanding","date":"2022-07-19","arxiv_id":"2207.09514","repositories_listed":1,"syntology":null},{"url":"/paper/mm-alt-a-multimodal-automatic-lyric","slug":"mm-alt-a-multimodal-automatic-lyric","title":"MM-ALT: A Multimodal Automatic Lyric Transcription System","date":"2022-07-13","arxiv_id":"2207.06127","repositories_listed":1,"syntology":null},{"url":"/paper/speaker-anonymization-with-phonetic","slug":"speaker-anonymization-with-phonetic","title":"Speaker Anonymization with Phonetic Intermediate Representations","date":"2022-07-11","arxiv_id":"2207.04834","repositories_listed":1,"syntology":null},{"url":"/paper/behancepr-a-punctuation-restoration-dataset","slug":"behancepr-a-punctuation-restoration-dataset","title":"BehancePR: A Punctuation Restoration Dataset for Livestreaming Video Transcript","date":"2022-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/stop-a-dataset-for-spoken-task-oriented","slug":"stop-a-dataset-for-spoken-task-oriented","title":"STOP: A dataset for Spoken Task Oriented Semantic Parsing","date":"2022-06-29","arxiv_id":"2207.10643","repositories_listed":1,"syntology":null},{"url":"/paper/distilling-a-pretrained-language-model-to-a","slug":"distilling-a-pretrained-language-model-to-a","title":"Distilling a Pretrained Language Model to a Multilingual ASR Model","date":"2022-06-25","arxiv_id":"2206.12638","repositories_listed":1,"syntology":null},{"url":"/paper/a-systematic-comparison-of-phonetic-aware","slug":"a-systematic-comparison-of-phonetic-aware","title":"A Systematic Comparison of Phonetic Aware Techniques for Speech Enhancement","date":"2022-06-22","arxiv_id":"2206.11000","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-cross-domain-speech-recognition-with","slug":"boosting-cross-domain-speech-recognition-with","title":"Boosting Cross-Domain Speech Recognition with Self-Supervision","date":"2022-06-20","arxiv_id":"2206.09783","repositories_listed":1,"syntology":null},{"url":"/paper/avatar-unconstrained-audiovisual-speech","slug":"avatar-unconstrained-audiovisual-speech","title":"AVATAR: Unconstrained Audiovisual Speech Recognition","date":"2022-06-15","arxiv_id":"2206.07684","repositories_listed":1,"syntology":null},{"url":"/paper/lae-language-aware-encoder-for-monolingual","slug":"lae-language-aware-encoder-for-monolingual","title":"LAE: Language-Aware Encoder for Monolingual and Multilingual ASR","date":"2022-06-05","arxiv_id":"2206.02093","repositories_listed":1,"syntology":null},{"url":"/paper/ssr7000-a-synchronized-corpus-of-ultrasound","slug":"ssr7000-a-synchronized-corpus-of-ultrasound","title":"SSR7000: A Synchronized Corpus of Ultrasound Tongue Imaging for End-to-End Silent Speech Recognition","date":"2022-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fleurs-few-shot-learning-evaluation-of","slug":"fleurs-few-shot-learning-evaluation-of","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","date":"2022-05-25","arxiv_id":"2205.12446","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-with-image-descriptors-are","slug":"language-models-with-image-descriptors-are","title":"Language Models with Image Descriptors are Strong Few-Shot Video-Language Learners","date":"2022-05-22","arxiv_id":"2205.10747","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-with-image-descriptors-are#ran","syntology_url":"https://syntology.ai/paper/2205.10747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10747"}},"official":{"repos":["mikewangwzhl/vidil"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/who-are-we-talking-about-handling-person-2","slug":"who-are-we-talking-about-handling-person-2","title":"Who Are We Talking About? Handling Person Names in Speech Translation","date":"2022-05-13","arxiv_id":"2205.06755","repositories_listed":1,"syntology":null},{"url":"/paper/vietnamese-automatic-speech-recognition-using","slug":"vietnamese-automatic-speech-recognition-using","title":"Vietnamese Automatic Speech Recognition using Wav2vec 2.0","date":"2022-05-08","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-multi-aspect-multi","slug":"transformer-based-multi-aspect-multi","title":"Transformer-Based Multi-Aspect Multi-Granularity Non-Native English Speaker Pronunciation Assessment","date":"2022-05-06","arxiv_id":"2205.03432","repositories_listed":1,"syntology":null},{"url":"/paper/speaker-recognition-in-the-wild","slug":"speaker-recognition-in-the-wild","title":"Speaker Recognition in the Wild","date":"2022-05-05","arxiv_id":"2205.02475","repositories_listed":1,"syntology":null},{"url":"/paper/wav2seq-pre-training-speech-to-text-encoder","slug":"wav2seq-pre-training-speech-to-text-encoder","title":"Wav2Seq: Pre-training Speech-to-Text Encoder-Decoder Models Using Pseudo Languages","date":"2022-05-02","arxiv_id":"2205.01086","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-speech-recognition-and-query-by","slug":"automatic-speech-recognition-and-query-by","title":"Automatic Speech Recognition and Query By Example for Creole Languages Documentation","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/how-you-say-it-matters-measuring-the-impact","slug":"how-you-say-it-matters-measuring-the-impact","title":"How You Say It Matters: Measuring the Impact of Verbal Disfluency Tags on Automated Dementia Detection","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-non-autoregressive-generation-for","slug":"a-survey-on-non-autoregressive-generation-for","title":"A Survey on Non-Autoregressive Generation for Neural Machine Translation and Beyond","date":"2022-04-20","arxiv_id":"2204.09269","repositories_listed":1,"syntology":null},{"url":"/paper/hubert-ee-early-exiting-hubert-for-efficient","slug":"hubert-ee-early-exiting-hubert-for-efficient","title":"HuBERT-EE: Early Exiting HuBERT for Efficient Speech Recognition","date":"2022-04-13","arxiv_id":"2204.06328","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-streaming-end-to-end-speech","slug":"large-scale-streaming-end-to-end-speech","title":"Large-Scale Streaming End-to-End Speech Translation with Neural Transducers","date":"2022-04-11","arxiv_id":"2204.05352","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-scale-streaming-end-to-end-speech#ran","syntology_url":"https://syntology.ai/paper/2204.05352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.05352"}},"official":null}},{"url":"/paper/unsupervised-uncertainty-measures-of","slug":"unsupervised-uncertainty-measures-of","title":"Unsupervised Uncertainty Measures of Automatic Speech Recognition for Non-intrusive Speech Intelligibility Prediction","date":"2022-04-08","arxiv_id":"2204.04288","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsupervised-uncertainty-measures-of#ran","syntology_url":"https://syntology.ai/paper/2204.04288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.04288"}},"official":{"repos":["claritychallenge/clarity"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/combining-spectral-and-self-supervised","slug":"combining-spectral-and-self-supervised","title":"Combining Spectral and Self-Supervised Features for Low Resource Speech Recognition and Translation","date":"2022-04-05","arxiv_id":"2204.02470","repositories_listed":1,"syntology":null},{"url":"/paper/towards-end-to-end-unsupervised-speech","slug":"towards-end-to-end-unsupervised-speech","title":"Towards End-to-end Unsupervised Speech Recognition","date":"2022-04-05","arxiv_id":"2204.02492","repositories_listed":1,"syntology":null},{"url":"/paper/primock57-a-dataset-of-primary-care-mock","slug":"primock57-a-dataset-of-primary-care-mock","title":"PriMock57: A Dataset Of Primary Care Mock Consultations","date":"2022-04-01","arxiv_id":"2204.00333","repositories_listed":1,"syntology":null},{"url":"/paper/a-hybrid-continuity-loss-to-reduce-over","slug":"a-hybrid-continuity-loss-to-reduce-over","title":"A Hybrid Continuity Loss to Reduce Over-Suppression for Time-domain Target Speaker Extraction","date":"2022-03-31","arxiv_id":"2203.16843","repositories_listed":1,"syntology":null},{"url":"/paper/indic-punct-an-automatic-punctuation","slug":"indic-punct-an-automatic-punctuation","title":"indic-punct: An automatic punctuation restoration and inverse text normalization framework for Indic languages","date":"2022-03-31","arxiv_id":"2203.16825","repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-transformer-decoder-for-end-to","slug":"pre-training-transformer-decoder-for-end-to","title":"Pre-Training Transformer Decoder for End-to-End ASR Model with Unpaired Speech Data","date":"2022-03-31","arxiv_id":"2203.17113","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-speaker-attributed-asr-with-token","slug":"streaming-speaker-attributed-asr-with-token","title":"Streaming Speaker-Attributed ASR with Token-Level Speaker Embeddings","date":"2022-03-30","arxiv_id":"2203.16685","repositories_listed":1,"syntology":null},{"url":"/paper/using-adapters-to-overcome-catastrophic","slug":"using-adapters-to-overcome-catastrophic","title":"Using Adapters to Overcome Catastrophic Forgetting in End-to-End Automatic Speech Recognition","date":"2022-03-30","arxiv_id":"2203.16082","repositories_listed":1,"syntology":null},{"url":"/paper/4-bit-conformer-with-native-quantization","slug":"4-bit-conformer-with-native-quantization","title":"4-bit Conformer with Native Quantization Aware Training for Speech Recognition","date":"2022-03-29","arxiv_id":"2203.15952","repositories_listed":1,"syntology":null},{"url":"/paper/a-single-speaker-is-almost-all-you-need-for","slug":"a-single-speaker-is-almost-all-you-need-for","title":"ASR data augmentation in low-resource settings using cross-lingual multi-speaker TTS and cross-lingual voice conversion","date":"2022-03-29","arxiv_id":"2204.00618","repositories_listed":1,"syntology":null},{"url":"/paper/analysis-of-eeg-frequency-bands-for","slug":"analysis-of-eeg-frequency-bands-for","title":"Analysis of EEG frequency bands for Envisioned Speech Recognition","date":"2022-03-29","arxiv_id":"2203.15250","repositories_listed":1,"syntology":null},{"url":"/paper/earnings-22-a-practical-benchmark-for-accents","slug":"earnings-22-a-practical-benchmark-for-accents","title":"Earnings-22: A Practical Benchmark for Accents in the Wild","date":"2022-03-29","arxiv_id":"2203.15591","repositories_listed":1,"syntology":null},{"url":"/paper/integrate-lattice-free-mmi-into-end-to-end","slug":"integrate-lattice-free-mmi-into-end-to-end","title":"Integrating Lattice-Free MMI into End-to-End Speech Recognition","date":"2022-03-29","arxiv_id":"2203.15614","repositories_listed":1,"syntology":null},{"url":"/paper/lighthubert-lightweight-and-configurable","slug":"lighthubert-lightweight-and-configurable","title":"LightHuBERT: Lightweight and Configurable Speech Representation Learning with Once-for-All Hidden-Unit BERT","date":"2022-03-29","arxiv_id":"2203.15610","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/lighthubert-lightweight-and-configurable#ran","syntology_url":"https://syntology.ai/paper/2203.15610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15610"}},"official":{"repos":["mechanicalsea/lighthubert"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/shifted-chunk-encoder-for-transformer-based","slug":"shifted-chunk-encoder-for-transformer-based","title":"Shifted Chunk Encoder for Transformer Based Streaming End-to-End ASR","date":"2022-03-29","arxiv_id":"2203.15206","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-text-to-speech-synthesis-by","slug":"unsupervised-text-to-speech-synthesis-by","title":"Unsupervised Text-to-Speech Synthesis by Unsupervised Automatic Speech Recognition","date":"2022-03-29","arxiv_id":"2203.15796","repositories_listed":1,"syntology":null},{"url":"/paper/cmgan-conformer-based-metric-gan-for-speech","slug":"cmgan-conformer-based-metric-gan-for-speech","title":"CMGAN: Conformer-based Metric GAN for Speech Enhancement","date":"2022-03-28","arxiv_id":"2203.15149","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":2,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 3 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cmgan-conformer-based-metric-gan-for-speech#ran","syntology_url":"https://syntology.ai/paper/2203.15149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15149"}},"official":{"repos":["ruizhecao96/cmgan"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-path-style-learning-for-end-to-end-noise","slug":"dual-path-style-learning-for-end-to-end-noise","title":"Dual-Path Style Learning for End-to-End Noise-Robust Speech Recognition","date":"2022-03-28","arxiv_id":"2203.14838","repositories_listed":1,"syntology":null},{"url":"/paper/finnish-parliament-asr-corpus-analysis","slug":"finnish-parliament-asr-corpus-analysis","title":"Finnish Parliament ASR corpus - Analysis, benchmarks and statistics","date":"2022-03-28","arxiv_id":"2203.14876","repositories_listed":1,"syntology":null},{"url":"/paper/a-dataset-for-speech-emotion-recognition-in","slug":"a-dataset-for-speech-emotion-recognition-in","title":"A Dataset for Speech Emotion Recognition in Greek Theatrical Plays","date":"2022-03-27","arxiv_id":"2203.15568","repositories_listed":1,"syntology":null},{"url":"/paper/towards-privacy-preserving-speech","slug":"towards-privacy-preserving-speech","title":"A Speech Representation Anonymization Framework via Selective Noise Perturbation","date":"2022-03-26","arxiv_id":"2203.14171","repositories_listed":1,"syntology":null},{"url":"/paper/speech-enhanced-and-noise-aware-networks-for","slug":"speech-enhanced-and-noise-aware-networks-for","title":"Speech-enhanced and Noise-aware Networks for Robust Speech Recognition","date":"2022-03-25","arxiv_id":"2203.13696","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-speech-recognition-for-speech","slug":"automatic-speech-recognition-for-speech","title":"Automatic Speech Recognition for Speech Assessment of Persian Preschool Children","date":"2022-03-24","arxiv_id":"2203.12886","repositories_listed":1,"syntology":null},{"url":"/paper/neural-predictor-for-black-box-adversarial","slug":"neural-predictor-for-black-box-adversarial","title":"Neural Predictor for Black-Box Adversarial Attacks on Speech Recognition","date":"2022-03-18","arxiv_id":"2203.09849","repositories_listed":1,"syntology":null},{"url":"/paper/red-ace-robust-error-detection-for-asr-using-1","slug":"red-ace-robust-error-detection-for-asr-using-1","title":"RED-ACE: Robust Error Detection for ASR using Confidence Embeddings","date":"2022-03-14","arxiv_id":"2203.07172","repositories_listed":1,"syntology":null},{"url":"/paper/dual-textless-spoken-question-answering-with-1","slug":"dual-textless-spoken-question-answering-with-1","title":"DUAL: Discrete Spoken Unit Adaptive Learning for Textless Spoken Question Answering","date":"2022-03-09","arxiv_id":"2203.04911","repositories_listed":1,"syntology":null},{"url":"/paper/towards-contextual-spelling-correction-for","slug":"towards-contextual-spelling-correction-for","title":"Towards Contextual Spelling Correction for Customization of End-to-end Speech Recognition Systems","date":"2022-03-02","arxiv_id":"2203.00888","repositories_listed":1,"syntology":null},{"url":"/paper/sentiment-word-aware-multimodal-refinement","slug":"sentiment-word-aware-multimodal-refinement","title":"Sentiment Word Aware Multimodal Refinement for Multimodal Sentiment Analysis with ASR Errors","date":"2022-03-01","arxiv_id":"2203.00257","repositories_listed":1,"syntology":null},{"url":"/paper/improving-ctc-based-speech-recognition-via","slug":"improving-ctc-based-speech-recognition-via","title":"Improving CTC-based speech recognition via knowledge transferring from pre-trained language models","date":"2022-02-22","arxiv_id":"2203.03582","repositories_listed":1,"syntology":null},{"url":"/paper/aishell-ner-named-entity-recognition-from","slug":"aishell-ner-named-entity-recognition-from","title":"AISHELL-NER: Named Entity Recognition from Chinese Speech","date":"2022-02-17","arxiv_id":"2202.08533","repositories_listed":1,"syntology":null},{"url":"/paper/adima-abuse-detection-in-multilingual-audio","slug":"adima-abuse-detection-in-multilingual-audio","title":"ADIMA: Abuse Detection In Multilingual Audio","date":"2022-02-16","arxiv_id":"2202.07991","repositories_listed":1,"syntology":null},{"url":"/paper/improving-automatic-speech-recognition-for","slug":"improving-automatic-speech-recognition-for","title":"Improving Automatic Speech Recognition for Non-Native English with Transfer Learning and Language Model Decoding","date":"2022-02-10","arxiv_id":"2202.05209","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-adapter-transfer-of-self-supervised","slug":"efficient-adapter-transfer-of-self-supervised","title":"Efficient Adapter Transfer of Self-Supervised Speech Models for Automatic Speech Recognition","date":"2022-02-07","arxiv_id":"2202.03218","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/efficient-adapter-transfer-of-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2202.03218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03218"}},"official":null}},{"url":"/paper/streaming-multi-talker-asr-with-token-level","slug":"streaming-multi-talker-asr-with-token-level","title":"Streaming Multi-Talker ASR with Token-Level Serialized Output Training","date":"2022-02-02","arxiv_id":"2202.00842","repositories_listed":1,"syntology":null},{"url":"/paper/star-temporal-classification-sequence","slug":"star-temporal-classification-sequence","title":"Star Temporal Classification: Sequence Classification with Partially Labeled Data","date":"2022-01-28","arxiv_id":"2201.12208","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-phonetic-inventories-with","slug":"discovering-phonetic-inventories-with","title":"Discovering Phonetic Inventories with Crosslingual Automatic Speech Recognition","date":"2022-01-26","arxiv_id":"2201.11207","repositories_listed":1,"syntology":null},{"url":"/paper/unified-multimodal-punctuation-restoration","slug":"unified-multimodal-punctuation-restoration","title":"Unified Multimodal Punctuation Restoration Framework for Mixed-Modality Corpus","date":"2022-01-24","arxiv_id":"2202.00468","repositories_listed":1,"syntology":null},{"url":"/paper/neural-architecture-search-for-lf-mmi-trained","slug":"neural-architecture-search-for-lf-mmi-trained","title":"Neural Architecture Search For LF-MMI Trained Time Delay Neural Networks","date":"2022-01-08","arxiv_id":"2201.03943","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-speech-recognition-datasets-in","slug":"automatic-speech-recognition-datasets-in","title":"Automatic Speech Recognition Datasets in Cantonese: A Survey and New Dataset","date":"2022-01-07","arxiv_id":"2201.02419","repositories_listed":1,"syntology":null},{"url":"/paper/improving-mandarin-end-to-end-speech","slug":"improving-mandarin-end-to-end-speech","title":"Improving Mandarin End-to-End Speech Recognition with Word N-gram Language Model","date":"2022-01-06","arxiv_id":"2201.01995","repositories_listed":1,"syntology":null},{"url":"/paper/robust-self-supervised-audio-visual-speech","slug":"robust-self-supervised-audio-visual-speech","title":"Robust Self-Supervised Audio-Visual Speech Recognition","date":"2022-01-05","arxiv_id":"2201.01763","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-self-supervised-audio-visual-speech#ran","syntology_url":"https://syntology.ai/paper/2201.01763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.01763"}},"official":{"repos":["facebookresearch/av_hubert"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/regularizing-end-to-end-speech-translation","slug":"regularizing-end-to-end-speech-translation","title":"Regularizing End-to-End Speech Translation with Triangular Decomposition Agreement","date":"2021-12-21","arxiv_id":"2112.10991","repositories_listed":1,"syntology":null},{"url":"/paper/continual-learning-for-monolingual-end-to-end","slug":"continual-learning-for-monolingual-end-to-end","title":"Continual Learning for Monolingual End-to-End Automatic Speech Recognition","date":"2021-12-17","arxiv_id":"2112.09427","repositories_listed":1,"syntology":null},{"url":"/paper/x-vector-based-voice-activity-detection-for-1","slug":"x-vector-based-voice-activity-detection-for-1","title":"X-Vector based voice activity detection for multi-genre broadcast speech-to-text","date":"2021-12-09","arxiv_id":"2112.05016","repositories_listed":1,"syntology":null},{"url":"/paper/consistent-training-and-decoding-for-end-to","slug":"consistent-training-and-decoding-for-end-to","title":"Consistent Training and Decoding For End-to-end Speech Recognition Using Lattice-free MMI","date":"2021-12-05","arxiv_id":"2112.02498","repositories_listed":1,"syntology":null},{"url":"/paper/slue-new-benchmark-tasks-for-spoken-language","slug":"slue-new-benchmark-tasks-for-spoken-language","title":"SLUE: New Benchmark Tasks for Spoken Language Understanding Evaluation on Natural Speech","date":"2021-11-19","arxiv_id":"2111.10367","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/slue-new-benchmark-tasks-for-spoken-language#ran","syntology_url":"https://syntology.ai/paper/2111.10367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.10367"}},"official":{"repos":["asappresearch/slue-toolkit"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sequential-randomized-smoothing-for-1","slug":"sequential-randomized-smoothing-for-1","title":"Sequential Randomized Smoothing for Adversarially Robust Speech Recognition","date":"2021-11-05","arxiv_id":"2112.03000","repositories_listed":1,"syntology":null},{"url":"/paper/a-transfer-learning-based-approach-for","slug":"a-transfer-learning-based-approach-for","title":"A transfer learning based approach for pronunciation scoring","date":"2021-11-01","arxiv_id":"2111.00976","repositories_listed":1,"syntology":null},{"url":"/paper/cross-attention-augmented-transducer-networks","slug":"cross-attention-augmented-transducer-networks","title":"Cross Attention Augmented Transducer Networks for Simultaneous Translation","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/intrinsic-evaluation-of-language-models-for","slug":"intrinsic-evaluation-of-language-models-for","title":"Intrinsic evaluation of language models for code-switching","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/revealing-and-protecting-labels-in","slug":"revealing-and-protecting-labels-in","title":"Revealing and Protecting Labels in Distributed Training","date":"2021-10-31","arxiv_id":"2111.00556","repositories_listed":1,"syntology":null},{"url":"/paper/aequevox-automated-fairness-testing-of-speech","slug":"aequevox-automated-fairness-testing-of-speech","title":"AequeVox: Automated Fairness Testing of Speech Recognition Systems","date":"2021-10-19","arxiv_id":"2110.09843","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-speaker-adaptation-approach-for-asr","slug":"a-unified-speaker-adaptation-approach-for-asr","title":"A Unified Speaker Adaptation Approach for ASR","date":"2021-10-16","arxiv_id":"2110.08545","repositories_listed":1,"syntology":null},{"url":"/paper/identifying-introductions-in-podcast-episodes","slug":"identifying-introductions-in-podcast-episodes","title":"Identifying Introductions in Podcast Episodes from Automatically Generated Transcripts","date":"2021-10-14","arxiv_id":"2110.07096","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-conditional-end-to-end-asr-with","slug":"hierarchical-conditional-end-to-end-asr-with","title":"Hierarchical Conditional End-to-End ASR with CTC and Multi-Granular Subword Units","date":"2021-10-08","arxiv_id":"2110.04109","repositories_listed":1,"syntology":null},{"url":"/paper/disambiguation-bert-for-n-best-rescoring-in","slug":"disambiguation-bert-for-n-best-rescoring-in","title":"BERT Attends the Conversation: Improving Low-Resource Conversational ASR","date":"2021-10-05","arxiv_id":"2110.02267","repositories_listed":1,"syntology":null},{"url":"/paper/fastcorrect-2-fast-error-correction-on","slug":"fastcorrect-2-fast-error-correction-on","title":"FastCorrect 2: Fast Error Correction on Multiple Candidates for Automatic Speech Recognition","date":"2021-09-29","arxiv_id":"2109.14420","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fastcorrect-2-fast-error-correction-on#ran","syntology_url":"https://syntology.ai/paper/2109.14420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.14420"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/factorized-neural-transducer-for-efficient","slug":"factorized-neural-transducer-for-efficient","title":"Factorized Neural Transducer for Efficient Language Model Adaptation","date":"2021-09-27","arxiv_id":"2110.01500","repositories_listed":1,"syntology":null},{"url":"/paper/fast-md-fast-multi-decoder-end-to-end-speech","slug":"fast-md-fast-multi-decoder-end-to-end-speech","title":"Fast-MD: Fast Multi-Decoder End-to-End Speech Translation with Non-Autoregressive Hidden Intermediates","date":"2021-09-27","arxiv_id":"2109.12804","repositories_listed":1,"syntology":null},{"url":"/paper/performance-efficiency-trade-offs-in","slug":"performance-efficiency-trade-offs-in","title":"Performance-Efficiency Trade-offs in Unsupervised Pre-training for Speech Recognition","date":"2021-09-14","arxiv_id":"2109.06870","repositories_listed":1,"syntology":null},{"url":"/paper/multi-sentence-resampling-a-simple-approach","slug":"multi-sentence-resampling-a-simple-approach","title":"Multi-Sentence Resampling: A Simple Approach to Alleviate Dataset Length Bias and Beam-Search Degradation","date":"2021-09-13","arxiv_id":"2109.06253","repositories_listed":1,"syntology":null}],"record_sha256":"6111a758c693732faae043f9825d3b9f94216d2f15ac8d57c9e34a1b4f9fc89d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}