{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition-2/papers/3","list_of":"/task/automatic-speech-recognition-2","task":"Automatic Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":32,"rows_per_page":100,"rows":[201,300],"of":3174,"counts":{"archive_papers_tagged":3174,"with_a_code_link":677,"where_syntology_ran_a_sample":79,"not_listed_spam_title":0,"listed":3174,"listed_where_code_ran":79,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":62,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":62,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition-2","prev":"/task/automatic-speech-recognition-2/papers/2","next":"/task/automatic-speech-recognition-2/papers/4","papers":[{"url":"/paper/arzen-llm-code-switched-egyptian-arabic","slug":"arzen-llm-code-switched-egyptian-arabic","title":"ArzEn-LLM: Code-Switched Egyptian Arabic-English Translation and Speech Recognition Using LLMs","date":"2024-06-26","arxiv_id":"2406.18120","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-speech-recognition-for-the-nepali","slug":"automatic-speech-recognition-for-the-nepali","title":"Automatic speech recognition for the Nepali language using CNN, bidirectional LSTM and ResNet","date":"2024-06-25","arxiv_id":"2406.17825","repositories_listed":1,"syntology":null},{"url":"/paper/fasa-a-flexible-and-automatic-speech-aligner","slug":"fasa-a-flexible-and-automatic-speech-aligner","title":"FASA: a Flexible and Automatic Speech Aligner for Extracting High-quality Aligned Children Speech Data","date":"2024-06-25","arxiv_id":"2406.17926","repositories_listed":1,"syntology":null},{"url":"/paper/towards-building-an-end-to-end-multilingual","slug":"towards-building-an-end-to-end-multilingual","title":"Towards Building an End-to-End Multilingual Automatic Lyrics Transcription Model","date":"2024-06-25","arxiv_id":"2406.17618","repositories_listed":1,"syntology":null},{"url":"/paper/growing-trees-on-sounds-assessing-strategies","slug":"growing-trees-on-sounds-assessing-strategies","title":"Growing Trees on Sounds: Assessing Strategies for End-to-End Dependency Parsing of Speech","date":"2024-06-18","arxiv_id":"2406.12621","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-online-continual-learning-for","slug":"unsupervised-online-continual-learning-for","title":"Unsupervised Online Continual Learning for Automatic Speech Recognition","date":"2024-06-18","arxiv_id":"2406.12503","repositories_listed":1,"syntology":null},{"url":"/paper/continual-test-time-adaptation-for-end-to-end","slug":"continual-test-time-adaptation-for-end-to-end","title":"Continual Test-time Adaptation for End-to-end Speech Recognition on Noisy Speech","date":"2024-06-16","arxiv_id":"2406.11064","repositories_listed":1,"syntology":null},{"url":"/paper/optimized-speculative-sampling-for-gpu","slug":"optimized-speculative-sampling-for-gpu","title":"Optimized Speculative Sampling for GPU Hardware Accelerators","date":"2024-06-16","arxiv_id":"2406.11016","repositories_listed":1,"syntology":null},{"url":"/paper/language-complexity-and-speech-recognition","slug":"language-complexity-and-speech-recognition","title":"Language Complexity and Speech Recognition Accuracy: Orthographic Complexity Hurts, Phonological Complexity Doesn't","date":"2024-06-13","arxiv_id":"2406.09202","repositories_listed":1,"syntology":null},{"url":"/paper/laser-learning-by-aligning-self-supervised","slug":"laser-learning-by-aligning-self-supervised","title":"LASER: Learning by Aligning Self-supervised Representations of Speech for Improving Content-related Tasks","date":"2024-06-13","arxiv_id":"2406.09153","repositories_listed":1,"syntology":null},{"url":"/paper/guiding-frame-level-ctc-alignments-using-self","slug":"guiding-frame-level-ctc-alignments-using-self","title":"Guiding Frame-Level CTC Alignments Using Self-knowledge Distillation","date":"2024-06-12","arxiv_id":"2406.07909","repositories_listed":1,"syntology":null},{"url":"/paper/speech-emotion-recognition-with-asr","slug":"speech-emotion-recognition-with-asr","title":"Speech Emotion Recognition with ASR Transcripts: A Comprehensive Study on Word Error Rate and Fusion Techniques","date":"2024-06-12","arxiv_id":"2406.08353","repositories_listed":1,"syntology":null},{"url":"/paper/towards-unsupervised-speech-recognition","slug":"towards-unsupervised-speech-recognition","title":"Towards Unsupervised Speech Recognition Without Pronunciation Models","date":"2024-06-12","arxiv_id":"2406.08380","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-unsupervised-speech-recognition#ran","syntology_url":"https://syntology.ai/paper/2406.08380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08380"}},"official":{"repos":["jeromeni/wholeword-uasr-jstti"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lipger-visually-conditioned-generative-error","slug":"lipger-visually-conditioned-generative-error","title":"LipGER: Visually-Conditioned Generative Error Correction for Robust Automatic Speech Recognition","date":"2024-06-06","arxiv_id":"2406.04432","repositories_listed":1,"syntology":null},{"url":"/paper/to-distill-or-not-to-distill-on-the","slug":"to-distill-or-not-to-distill-on-the","title":"To Distill or Not to Distill? On the Robustness of Robust Knowledge Distillation","date":"2024-06-06","arxiv_id":"2406.04512","repositories_listed":1,"syntology":null},{"url":"/paper/error-preserving-automatic-speech-recognition","slug":"error-preserving-automatic-speech-recognition","title":"Error-preserving Automatic Speech Recognition of Young English Learners' Language","date":"2024-06-05","arxiv_id":"2406.03235","repositories_listed":1,"syntology":null},{"url":"/paper/whistle-data-efficient-multilingual-and","slug":"whistle-data-efficient-multilingual-and","title":"Whistle: Data-Efficient Multilingual and Crosslingual Speech Recognition via Weakly Phonetic Supervision","date":"2024-06-04","arxiv_id":"2406.02166","repositories_listed":1,"syntology":null},{"url":"/paper/a-variance-preserving-interpolation-approach","slug":"a-variance-preserving-interpolation-approach","title":"A Variance-Preserving Interpolation Approach for Diffusion Models with Applications to Single Channel Speech Enhancement and Recognition","date":"2024-05-27","arxiv_id":"2405.16952","repositories_listed":1,"syntology":null},{"url":"/paper/federating-dynamic-models-using-early-exit","slug":"federating-dynamic-models-using-early-exit","title":"Federating Dynamic Models using Early-Exit Architectures for Automatic Speech Recognition on Heterogeneous Clients","date":"2024-05-27","arxiv_id":"2405.17376","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-and-consistency-learning-for","slug":"contrastive-and-consistency-learning-for","title":"Contrastive and Consistency Learning for Neural Noisy-Channel Model in Spoken Language Understanding","date":"2024-05-23","arxiv_id":"2405.15097","repositories_listed":1,"syntology":null},{"url":"/paper/let-s-fuse-step-by-step-a-generative-fusion","slug":"let-s-fuse-step-by-step-a-generative-fusion","title":"Let's Fuse Step by Step: A Generative Fusion Decoding Algorithm with LLMs for Multi-modal Text Recognition","date":"2024-05-23","arxiv_id":"2405.14259","repositories_listed":1,"syntology":null},{"url":"/paper/self-taught-recognizer-toward-unsupervised","slug":"self-taught-recognizer-toward-unsupervised","title":"Self-Taught Recognizer: Toward Unsupervised Adaptation for Speech Foundation Models","date":"2024-05-23","arxiv_id":"2405.14161","repositories_listed":1,"syntology":null},{"url":"/paper/soccernet-echoes-a-soccer-game-audio","slug":"soccernet-echoes-a-soccer-game-audio","title":"SoccerNet-Echoes: A Soccer Game Audio Commentary Dataset","date":"2024-05-12","arxiv_id":"2405.07354","repositories_listed":1,"syntology":null},{"url":"/paper/muting-whisper-a-universal-acoustic","slug":"muting-whisper-a-universal-acoustic","title":"Muting Whisper: A Universal Acoustic Adversarial Attack on Speech Foundation Models","date":"2024-05-09","arxiv_id":"2405.06134","repositories_listed":1,"syntology":null},{"url":"/paper/open-implementation-and-study-of-best-rq-for","slug":"open-implementation-and-study-of-best-rq-for","title":"Open Implementation and Study of BEST-RQ for Speech Processing","date":"2024-05-07","arxiv_id":"2405.04296","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-the-potential-of-llm-based-asr-on","slug":"unveiling-the-potential-of-llm-based-asr-on","title":"Unveiling the Potential of LLM-Based ASR on Chinese Open-Source Datasets","date":"2024-05-03","arxiv_id":"2405.02132","repositories_listed":1,"syntology":null},{"url":"/paper/killkan-the-automatic-speech-recognition","slug":"killkan-the-automatic-speech-recognition","title":"Killkan: The Automatic Speech Recognition Dataset for Kichwa with Morphosyntactic Information","date":"2024-04-23","arxiv_id":"2404.15501","repositories_listed":1,"syntology":null},{"url":"/paper/less-peaky-and-more-accurate-ctc-forced","slug":"less-peaky-and-more-accurate-ctc-forced","title":"Less Peaky and More Accurate CTC Forced Alignment by Label Priors","date":"2024-04-22","arxiv_id":"2406.02560","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/less-peaky-and-more-accurate-ctc-forced#ran","syntology_url":"https://syntology.ai/paper/2406.02560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.02560"}},"official":{"repos":["huangruizhe/audio"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/semantically-corrected-amharic-automatic","slug":"semantically-corrected-amharic-automatic","title":"Semantically Corrected Amharic Automatic Speech Recognition","date":"2024-04-20","arxiv_id":"2404.13362","repositories_listed":1,"syntology":null},{"url":"/paper/kallaama-a-transcribed-speech-dataset-about","slug":"kallaama-a-transcribed-speech-dataset-about","title":"Kallaama: A Transcribed Speech Dataset about Agriculture in the Three Most Widely Spoken Languages in Senegal","date":"2024-04-02","arxiv_id":"2404.01991","repositories_listed":1,"syntology":null},{"url":"/paper/elitr-bench-a-meeting-assistant-benchmark-for","slug":"elitr-bench-a-meeting-assistant-benchmark-for","title":"ELITR-Bench: A Meeting Assistant Benchmark for Long-Context Language Models","date":"2024-03-29","arxiv_id":"2403.20262","repositories_listed":1,"syntology":null},{"url":"/paper/phowhisper-automatic-speech-recognition-for","slug":"phowhisper-automatic-speech-recognition-for","title":"PhoWhisper: Automatic Speech Recognition for Vietnamese","date":"2024-03-27","arxiv_id":"2406.02555","repositories_listed":1,"syntology":null},{"url":"/paper/speechcolab-leaderboard-an-open-source","slug":"speechcolab-leaderboard-an-open-source","title":"SpeechColab Leaderboard: An Open-Source Platform for Automatic Speech Recognition Evaluation","date":"2024-03-13","arxiv_id":"2403.08196","repositories_listed":1,"syntology":null},{"url":"/paper/score-self-supervised-correspondence-fine","slug":"score-self-supervised-correspondence-fine","title":"SCORE: Self-supervised Correspondence Fine-tuning for Improved Content Representations","date":"2024-03-10","arxiv_id":"2403.06260","repositories_listed":1,"syntology":null},{"url":"/paper/speech-robust-bench-a-robustness-benchmark","slug":"speech-robust-bench-a-robustness-benchmark","title":"Speech Robust Bench: A Robustness Benchmark For Speech Recognition","date":"2024-03-08","arxiv_id":"2403.07937","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/speech-robust-bench-a-robustness-benchmark#ran","syntology_url":"https://syntology.ai/paper/2403.07937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07937"}},"official":{"repos":["ahmedshah1494/speech_robust_bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/language-and-speech-technology-for-central","slug":"language-and-speech-technology-for-central","title":"Language and Speech Technology for Central Kurdish Varieties","date":"2024-03-04","arxiv_id":"2403.01983","repositories_listed":1,"syntology":null},{"url":"/paper/pixit-joint-training-of-speaker-diarization","slug":"pixit-joint-training-of-speaker-diarization","title":"PixIT: Joint Training of Speaker Diarization and Speech Separation from Real-world Multi-speaker Recordings","date":"2024-03-04","arxiv_id":"2403.02288","repositories_listed":1,"syntology":null},{"url":"/paper/a-cross-modal-approach-to-silent-speech-with","slug":"a-cross-modal-approach-to-silent-speech-with","title":"A Cross-Modal Approach to Silent Speech with LLM-Enhanced Recognition","date":"2024-03-02","arxiv_id":"2403.05583","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-speech-models-for-automatic","slug":"multilingual-speech-models-for-automatic","title":"Twists, Humps, and Pebbles: Multilingual Speech Recognition Models Exhibit Gender Performance Gaps","date":"2024-02-28","arxiv_id":"2402.17954","repositories_listed":1,"syntology":null},{"url":"/paper/how-do-hyenas-deal-with-human-speech-speech","slug":"how-do-hyenas-deal-with-human-speech-speech","title":"How do Hyenas deal with Human Speech? Speech Recognition and Translation with ConfHyena","date":"2024-02-20","arxiv_id":"2402.13208","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-do-hyenas-deal-with-human-speech-speech#ran","syntology_url":"https://syntology.ai/paper/2402.13208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13208"}},"official":{"repos":["hlt-mt/fbk-fairseq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/owsm-ctc-an-open-encoder-only-speech","slug":"owsm-ctc-an-open-encoder-only-speech","title":"OWSM-CTC: An Open Encoder-Only Speech Foundation Model for Speech Recognition, Translation, and Language Identification","date":"2024-02-20","arxiv_id":"2402.12654","repositories_listed":1,"syntology":null},{"url":"/paper/air-bench-benchmarking-large-audio-language","slug":"air-bench-benchmarking-large-audio-language","title":"AIR-Bench: Benchmarking Large Audio-Language Models via Generative Comprehension","date":"2024-02-12","arxiv_id":"2402.07729","repositories_listed":1,"syntology":null},{"url":"/paper/it-s-never-too-late-fusing-acoustic","slug":"it-s-never-too-late-fusing-acoustic","title":"It's Never Too Late: Fusing Acoustic Information into Large Language Models for Automatic Speech Recognition","date":"2024-02-08","arxiv_id":"2402.05457","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/it-s-never-too-late-fusing-acoustic#ran","syntology_url":"https://syntology.ai/paper/2402.05457","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05457"}},"official":null}},{"url":"/paper/unified-speech-text-pretraining-for-spoken","slug":"unified-speech-text-pretraining-for-spoken","title":"Paralinguistics-Aware Speech-Empowered Large Language Models for Natural Conversation","date":"2024-02-08","arxiv_id":"2402.05706","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":5,"n_ran_checked":7,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unified-speech-text-pretraining-for-spoken#ran","syntology_url":"https://syntology.ai/paper/2402.05706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05706"}},"official":{"repos":["naver-ai/usdm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/reborn-reinforcement-learned-boundary","slug":"reborn-reinforcement-learned-boundary","title":"REBORN: Reinforcement-Learned Boundary Segmentation with Iterative Training for Unsupervised ASR","date":"2024-02-06","arxiv_id":"2402.03988","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/reborn-reinforcement-learned-boundary#ran","syntology_url":"https://syntology.ai/paper/2402.03988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03988"}},"official":{"repos":["andybi7676/reborn-uasr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/streaming-sequence-transduction-through","slug":"streaming-sequence-transduction-through","title":"Streaming Sequence Transduction through Dynamic Compression","date":"2024-02-02","arxiv_id":"2402.01172","repositories_listed":1,"syntology":null},{"url":"/paper/word-level-asr-quality-estimation-for","slug":"word-level-asr-quality-estimation-for","title":"Word-Level ASR Quality Estimation for Efficient Corpus Sampling and Post-Editing through Analyzing Attentions of a Reference-Free Metric","date":"2024-01-20","arxiv_id":"2401.11268","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-are-efficient-learners","slug":"large-language-models-are-efficient-learners","title":"Large Language Models are Efficient Learners of Noise-Robust Speech Recognition","date":"2024-01-19","arxiv_id":"2401.10446","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-are-efficient-learners#ran","syntology_url":"https://syntology.ai/paper/2401.10446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10446"}},"official":{"repos":["yuchen005/robustger"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cascaded-cross-modal-transformer-for-audio","slug":"cascaded-cross-modal-transformer-for-audio","title":"Cascaded Cross-Modal Transformer for Audio-Textual Classification","date":"2024-01-15","arxiv_id":"2401.07575","repositories_listed":1,"syntology":null},{"url":"/paper/multichannel-av-wav2vec2-a-framework-for","slug":"multichannel-av-wav2vec2-a-framework-for","title":"Multichannel AV-wav2vec2: A Framework for Learning Multichannel Multi-Modal Speech Representation","date":"2024-01-07","arxiv_id":"2401.03468","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":2,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"8 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multichannel-av-wav2vec2-a-framework-for#ran","syntology_url":"https://syntology.ai/paper/2401.03468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03468"}},"official":{"repos":["zqs01/multi-channel-wav2vec2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/teles-temporal-lexeme-similarity-score-to","slug":"teles-temporal-lexeme-similarity-score-to","title":"TeLeS: Temporal Lexeme Similarity Score to Estimate Confidence in End-to-End ASR","date":"2024-01-06","arxiv_id":"2401.03251","repositories_listed":1,"syntology":null},{"url":"/paper/task-oriented-dialogue-as-a-catalyst-for-self","slug":"task-oriented-dialogue-as-a-catalyst-for-self","title":"Task Oriented Dialogue as a Catalyst for Self-Supervised Automatic Speech Recognition","date":"2024-01-04","arxiv_id":"2401.02417","repositories_listed":1,"syntology":null},{"url":"/paper/stateful-fastconformer-with-cache-based","slug":"stateful-fastconformer-with-cache-based","title":"Stateful Conformer with Cache-based Inference for Streaming Automatic Speech Recognition","date":"2023-12-27","arxiv_id":"2312.17279","repositories_listed":1,"syntology":null},{"url":"/paper/stable-distillation-regularizing-continued","slug":"stable-distillation-regularizing-continued","title":"Stable Distillation: Regularizing Continued Pre-training for Low-Resource Automatic Speech Recognition","date":"2023-12-20","arxiv_id":"2312.12783","repositories_listed":1,"syntology":null},{"url":"/paper/seq2seq-for-automatic-paraphasia-detection-in","slug":"seq2seq-for-automatic-paraphasia-detection-in","title":"Seq2seq for Automatic Paraphasia Detection in Aphasic Speech","date":"2023-12-16","arxiv_id":"2312.10518","repositories_listed":1,"syntology":null},{"url":"/paper/extending-whisper-with-prompt-tuning-to","slug":"extending-whisper-with-prompt-tuning-to","title":"Extending Whisper with prompt tuning to target-speaker ASR","date":"2023-12-13","arxiv_id":"2312.08079","repositories_listed":1,"syntology":null},{"url":"/paper/rose-a-recognition-oriented-speech","slug":"rose-a-recognition-oriented-speech","title":"ROSE: A Recognition-Oriented Speech Enhancement Framework in Air Traffic Control Using Multi-Objective Learning","date":"2023-12-11","arxiv_id":"2312.06118","repositories_listed":1,"syntology":null},{"url":"/paper/bigger-is-not-always-better-the-effect-of","slug":"bigger-is-not-always-better-the-effect-of","title":"Bigger is not Always Better: The Effect of Context Size on Speech Pre-Training","date":"2023-12-03","arxiv_id":"2312.01515","repositories_listed":1,"syntology":null},{"url":"/paper/d4am-a-general-denoising-framework-for","slug":"d4am-a-general-denoising-framework-for","title":"D4AM: A General Denoising Framework for Downstream Acoustic Models","date":"2023-11-28","arxiv_id":"2311.16595","repositories_listed":1,"syntology":null},{"url":"/paper/a-quantitative-approach-to-understand-self","slug":"a-quantitative-approach-to-understand-self","title":"A Quantitative Approach to Understand Self-Supervised Models as Cross-lingual Feature Extractors","date":"2023-11-27","arxiv_id":"2311.15954","repositories_listed":1,"syntology":null},{"url":"/paper/lip-rtve-an-audiovisual-database-for-1","slug":"lip-rtve-an-audiovisual-database-for-1","title":"LIP-RTVE: An Audiovisual Database for Continuous Spanish in the Wild","date":"2023-11-21","arxiv_id":"2311.12457","repositories_listed":1,"syntology":null},{"url":"/paper/improving-whispered-speech-recognition","slug":"improving-whispered-speech-recognition","title":"Improving Whispered Speech Recognition Performance using Pseudo-whispered based Data Augmentation","date":"2023-11-09","arxiv_id":"2311.05179","repositories_listed":1,"syntology":null},{"url":"/paper/improved-child-text-to-speech-synthesis","slug":"improved-child-text-to-speech-synthesis","title":"Improved Child Text-to-Speech Synthesis through Fastpitch-based Transfer Learning","date":"2023-11-07","arxiv_id":"2311.04313","repositories_listed":1,"syntology":null},{"url":"/paper/pseudo-labeling-for-domain-agnostic-bangla","slug":"pseudo-labeling-for-domain-agnostic-bangla","title":"Pseudo-Labeling for Domain-Agnostic Bangla Automatic Speech Recognition","date":"2023-11-06","arxiv_id":"2311.03196","repositories_listed":1,"syntology":null},{"url":"/paper/distilwhisper-efficient-distillation-of-multi","slug":"distilwhisper-efficient-distillation-of-multi","title":"Multilingual DistilWhisper: Efficient Distillation of Multi-task Speech Models via Language-Specific Experts","date":"2023-11-02","arxiv_id":"2311.01070","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-single-channel-speaker-turn-aware","slug":"end-to-end-single-channel-speaker-turn-aware","title":"End-to-End Single-Channel Speaker-Turn Aware Conversational Speech Translation","date":"2023-11-01","arxiv_id":"2311.00697","repositories_listed":1,"syntology":null},{"url":"/paper/whisper-mce-whisper-model-finetuned-for","slug":"whisper-mce-whisper-model-finetuned-for","title":"Developing a Multilingual Dataset and Evaluation Metrics for Code-Switching: A Focus on Hong Kong's Polylingual Dynamics","date":"2023-10-27","arxiv_id":"2310.17953","repositories_listed":1,"syntology":null},{"url":"/paper/artst-arabic-text-and-speech-transformer","slug":"artst-arabic-text-and-speech-transformer","title":"ArTST: Arabic Text and Speech Transformer","date":"2023-10-25","arxiv_id":"2310.16621","repositories_listed":1,"syntology":null},{"url":"/paper/cl-masr-a-continual-learning-benchmark-for","slug":"cl-masr-a-continual-learning-benchmark-for","title":"CL-MASR: A Continual Learning Benchmark for Multilingual ASR","date":"2023-10-25","arxiv_id":"2310.16931","repositories_listed":1,"syntology":null},{"url":"/paper/disco-a-large-scale-human-annotated-corpus","slug":"disco-a-large-scale-human-annotated-corpus","title":"DISCO: A Large Scale Human Annotated Corpus for Disfluency Correction in Indo-European Languages","date":"2023-10-25","arxiv_id":"2310.16749","repositories_listed":1,"syntology":null},{"url":"/paper/accented-speech-recognition-with-accent","slug":"accented-speech-recognition-with-accent","title":"Accented Speech Recognition With Accent-specific Codebooks","date":"2023-10-24","arxiv_id":"2310.15970","repositories_listed":1,"syntology":null},{"url":"/paper/key-frame-mechanism-for-efficient-conformer","slug":"key-frame-mechanism-for-efficient-conformer","title":"Key Frame Mechanism For Efficient Conformer Based End-to-end Speech Recognition","date":"2023-10-23","arxiv_id":"2310.14954","repositories_listed":1,"syntology":null},{"url":"/paper/salmonn-towards-generic-hearing-abilities-for","slug":"salmonn-towards-generic-hearing-abilities-for","title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","date":"2023-10-20","arxiv_id":"2310.13289","repositories_listed":1,"syntology":null},{"url":"/paper/zipformer-a-faster-and-better-encoder-for","slug":"zipformer-a-faster-and-better-encoder-for","title":"Zipformer: A faster and better encoder for automatic speech recognition","date":"2023-10-17","arxiv_id":"2310.11230","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-test-time-adaptation-for-acoustic","slug":"advancing-test-time-adaptation-for-acoustic","title":"Advancing Test-Time Adaptation in Wild Acoustic Test Settings","date":"2023-10-14","arxiv_id":"2310.09505","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/advancing-test-time-adaptation-for-acoustic#ran","syntology_url":"https://syntology.ai/paper/2310.09505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09505"}},"official":{"repos":["Waffle-Liu/CEA"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/salm-speech-augmented-language-model-with-in","slug":"salm-speech-augmented-language-model-with-in","title":"SALM: Speech-augmented Language Model with In-context Learning for Speech Recognition and Translation","date":"2023-10-13","arxiv_id":"2310.09424","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-the-adapters-for-code-switching-in","slug":"adapting-the-adapters-for-code-switching-in","title":"Adapting the adapters for code-switching in multilingual ASR","date":"2023-10-11","arxiv_id":"2310.07423","repositories_listed":1,"syntology":null},{"url":"/paper/no-pitch-left-behind-addressing-gender","slug":"no-pitch-left-behind-addressing-gender","title":"No Pitch Left Behind: Addressing Gender Unbalance in Automatic Speech Recognition through Pitch Manipulation","date":"2023-10-10","arxiv_id":"2310.06590","repositories_listed":1,"syntology":null},{"url":"/paper/whispering-llama-a-cross-modal-generative","slug":"whispering-llama-a-cross-modal-generative","title":"Whispering LLaMA: A Cross-Modal Generative Error Correction Framework for Speech Recognition","date":"2023-10-10","arxiv_id":"2310.06434","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/whispering-llama-a-cross-modal-generative#ran","syntology_url":"https://syntology.ai/paper/2310.06434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06434"}},"official":{"repos":["srijith-rkr/whispering-llama"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ed-cec-improving-rare-word-recognition-using","slug":"ed-cec-improving-rare-word-recognition-using","title":"ed-cec: improving rare word recognition using asr postprocessing based on error detection and context-aware error correction","date":"2023-10-08","arxiv_id":"2310.05129","repositories_listed":1,"syntology":null},{"url":"/paper/howtocaption-prompting-llms-to-transform","slug":"howtocaption-prompting-llms-to-transform","title":"HowToCaption: Prompting LLMs to Transform Video Annotations at Scale","date":"2023-10-07","arxiv_id":"2310.04900","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/howtocaption-prompting-llms-to-transform#ran","syntology_url":"https://syntology.ai/paper/2310.04900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04900"}},"official":{"repos":["ninatu/howtocaption"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hyporadise-an-open-baseline-for-generative-1","slug":"hyporadise-an-open-baseline-for-generative-1","title":"HyPoradise: An Open Baseline for Generative Speech Recognition with Large Language Models","date":"2023-09-27","arxiv_id":"2309.15701","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hyporadise-an-open-baseline-for-generative-1#ran","syntology_url":"https://syntology.ai/paper/2309.15701","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15701"}},"official":{"repos":["hypotheses-paradise/hypo2trans"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/speech-collage-code-switched-audio-generation","slug":"speech-collage-code-switched-audio-generation","title":"Speech collage: code-switched audio generation by collaging monolingual corpora","date":"2023-09-27","arxiv_id":"2309.15674","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-flawed-data-weakly-supervised","slug":"learning-from-flawed-data-weakly-supervised","title":"Learning from Flawed Data: Weakly Supervised Automatic Speech Recognition","date":"2023-09-26","arxiv_id":"2309.15796","repositories_listed":1,"syntology":null},{"url":"/paper/segmentation-free-streaming-machine","slug":"segmentation-free-streaming-machine","title":"Segmentation-Free Streaming Machine Translation","date":"2023-09-26","arxiv_id":"2309.14823","repositories_listed":1,"syntology":null},{"url":"/paper/human-transcription-quality-improvement","slug":"human-transcription-quality-improvement","title":"Human Transcription Quality Improvement","date":"2023-09-24","arxiv_id":"2309.14372","repositories_listed":1,"syntology":null},{"url":"/paper/big-model-only-for-hard-audios-sample","slug":"big-model-only-for-hard-audios-sample","title":"Big model only for hard audios: Sample dependent Whisper model selection for efficient inferences","date":"2023-09-22","arxiv_id":"2309.12712","repositories_listed":1,"syntology":null},{"url":"/paper/memory-augmented-conformer-for-improved-end","slug":"memory-augmented-conformer-for-improved-end","title":"Memory-augmented conformer for improved end-to-end long-form ASR","date":"2023-09-22","arxiv_id":"2309.13029","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-the-zero-shot-power-of-instruction","slug":"harnessing-the-zero-shot-power-of-instruction","title":"Harnessing the Zero-Shot Power of Instruction-Tuned Large Language Model in End-to-End Speech Recognition","date":"2023-09-19","arxiv_id":"2309.10524","repositories_listed":1,"syntology":null},{"url":"/paper/hypr-a-comprehensive-study-for-asr-hypothesis","slug":"hypr-a-comprehensive-study-for-asr-hypothesis","title":"HypR: A comprehensive study for ASR hypothesis revising with a reference corpus","date":"2023-09-18","arxiv_id":"2309.09838","repositories_listed":1,"syntology":null},{"url":"/paper/training-dynamic-models-using-early-exits-for","slug":"training-dynamic-models-using-early-exits-for","title":"Training dynamic models using early exits for automatic speech recognition on resource-constrained devices","date":"2023-09-18","arxiv_id":"2309.09546","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-quantised-end-to-end-asr-models-via","slug":"enhancing-quantised-end-to-end-asr-models-via","title":"Enhancing Quantised End-to-End ASR Models via Personalisation","date":"2023-09-17","arxiv_id":"2309.09136","repositories_listed":1,"syntology":null},{"url":"/paper/diacorrect-error-correction-back-end-for","slug":"diacorrect-error-correction-back-end-for","title":"DiaCorrect: Error Correction Back-end For Speaker Diarization","date":"2023-09-15","arxiv_id":"2309.08377","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-punctuation-restoration-for","slug":"transformer-based-punctuation-restoration-for","title":"Transformer Based Punctuation Restoration for Turkish","date":"2023-09-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unimodal-aggregation-for-ctc-based-speech","slug":"unimodal-aggregation-for-ctc-based-speech","title":"Unimodal Aggregation for CTC-based Speech Recognition","date":"2023-09-15","arxiv_id":"2309.08150","repositories_listed":1,"syntology":null},{"url":"/paper/funcodec-a-fundamental-reproducible-and","slug":"funcodec-a-fundamental-reproducible-and","title":"FunCodec: A Fundamental, Reproducible and Integrable Open-source Toolkit for Neural Speech Codec","date":"2023-09-14","arxiv_id":"2309.07405","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-asr-for-resource-constrained-robots","slug":"hybrid-asr-for-resource-constrained-robots","title":"Hybrid ASR for Resource-Constrained Robots: HMM - Deep Learning Fusion","date":"2023-09-11","arxiv_id":"2309.07164","repositories_listed":1,"syntology":null},{"url":"/paper/perceptual-and-task-oriented-assessment-of-a","slug":"perceptual-and-task-oriented-assessment-of-a","title":"Perceptual and Task-Oriented Assessment of a Semantic Metric for ASR Evaluation","date":"2023-09-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-small-and-fast-bert-for-chinese-medical","slug":"a-small-and-fast-bert-for-chinese-medical","title":"A Small and Fast BERT for Chinese Medical Punctuation Restoration","date":"2023-08-24","arxiv_id":"2308.12568","repositories_listed":1,"syntology":null},{"url":"/paper/bayes-risk-transducer-transducer-with","slug":"bayes-risk-transducer-transducer-with","title":"Bayes Risk Transducer: Transducer with Controllable Alignment Prediction","date":"2023-08-19","arxiv_id":"2308.10107","repositories_listed":1,"syntology":null}],"record_sha256":"a4065a08d72826ae7b7d239143fdcbe0a333d14ea88445a36ed3e5a7cddb1a18","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}