{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/automatic-speech-recognition-2/papers/ran/1","list_of":"/task/automatic-speech-recognition-2","task":"Automatic Speech Recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,79],"of":79,"counts":{"archive_papers_tagged":3174,"with_a_code_link":677,"where_syntology_ran_a_sample":79,"not_listed_spam_title":0,"listed":3174,"listed_where_code_ran":79,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":62,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":62,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/automatic-speech-recognition-2/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/unifying-streaming-and-non-streaming","slug":"unifying-streaming-and-non-streaming","title":"Unifying Streaming and Non-streaming Zipformer-based ASR","date":"2025-06-17","arxiv_id":"2506.14434","repositories_listed":0,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unifying-streaming-and-non-streaming#ran","syntology_url":"https://syntology.ai/paper/2506.14434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.14434"}},"official":null}},{"url":"/paper/cosyvoice-3-towards-in-the-wild-speech","slug":"cosyvoice-3-towards-in-the-wild-speech","title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","date":"2025-05-23","arxiv_id":"2505.17589","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cosyvoice-3-towards-in-the-wild-speech#ran","syntology_url":"https://syntology.ai/paper/2505.17589","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17589"}},"official":{"repos":["funaudiollm/cosyvoice"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/daily-omni-towards-audio-visual-reasoning","slug":"daily-omni-towards-audio-visual-reasoning","title":"Daily-Omni: Towards Audio-Visual Reasoning with Temporal Alignment across Modalities","date":"2025-05-23","arxiv_id":"2505.17862","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/daily-omni-towards-audio-visual-reasoning#ran","syntology_url":"https://syntology.ai/paper/2505.17862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17862"}},"official":{"repos":["lliar-liar/daily-omni"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/voila-voice-language-foundation-models-for","slug":"voila-voice-language-foundation-models-for","title":"Voila: Voice-Language Foundation Models for Real-Time Autonomous Interaction and Voice Role-Play","date":"2025-05-05","arxiv_id":"2505.02707","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/voila-voice-language-foundation-models-for#ran","syntology_url":"https://syntology.ai/paper/2505.02707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02707"}},"official":{"repos":["maitrix-org/voila"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/whispering-under-the-eaves-protecting-user","slug":"whispering-under-the-eaves-protecting-user","title":"Whispering Under the Eaves: Protecting User Privacy Against Commercial and LLM-powered Automatic Speech Recognition Systems","date":"2025-04-01","arxiv_id":"2504.00858","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/whispering-under-the-eaves-protecting-user#ran","syntology_url":"https://syntology.ai/paper/2504.00858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.00858"}},"official":{"repos":["WeifeiJin/AudioShield"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/liteasr-efficient-automatic-speech","slug":"liteasr-efficient-automatic-speech","title":"LiteASR: Efficient Automatic Speech Recognition with Low-Rank Approximation","date":"2025-02-27","arxiv_id":"2502.20583","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/liteasr-efficient-automatic-speech#ran","syntology_url":"https://syntology.ai/paper/2502.20583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20583"}},"official":{"repos":["efeslab/liteasr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/glm-4-voice-towards-intelligent-and-human","slug":"glm-4-voice-towards-intelligent-and-human","title":"GLM-4-Voice: Towards Intelligent and Human-Like End-to-End Spoken Chatbot","date":"2024-12-03","arxiv_id":"2412.02612","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/glm-4-voice-towards-intelligent-and-human#ran","syntology_url":"https://syntology.ai/paper/2412.02612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02612"}},"official":{"repos":["thudm/glm-4-voice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/emg2qwerty-a-large-dataset-with-baselines-for","slug":"emg2qwerty-a-large-dataset-with-baselines-for","title":"emg2qwerty: A Large Dataset with Baselines for Touch Typing using Surface Electromyography","date":"2024-10-26","arxiv_id":"2410.20081","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emg2qwerty-a-large-dataset-with-baselines-for#ran","syntology_url":"https://syntology.ai/paper/2410.20081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20081"}},"official":{"repos":["facebookresearch/emg2qwerty"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/voicebench-benchmarking-llm-based-voice","slug":"voicebench-benchmarking-llm-based-voice","title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","date":"2024-10-22","arxiv_id":"2410.17196","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/voicebench-benchmarking-llm-based-voice#ran","syntology_url":"https://syntology.ai/paper/2410.17196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17196"}},"official":{"repos":["matthewcym/voicebench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-strong-audio-visual","slug":"large-language-models-are-strong-audio-visual","title":"Large Language Models are Strong Audio-Visual Speech Recognition Learners","date":"2024-09-18","arxiv_id":"2409.12319","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":2,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"9 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/large-language-models-are-strong-audio-visual#ran","syntology_url":"https://syntology.ai/paper/2409.12319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12319"}},"official":{"repos":["umbertocappellazzo/llama-avsr"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00005","slug":"2408-00005","title":"Framework for Curating Speech Datasets and Evaluating ASR Systems: A Case Study for Polish","date":"2024-07-18","arxiv_id":"2408.00005","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2408-00005#ran","syntology_url":"https://syntology.ai/paper/2408.00005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00005"}},"official":{"repos":["goodmike31/pl-asr-bigos-tools"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/towards-unsupervised-speech-recognition","slug":"towards-unsupervised-speech-recognition","title":"Towards Unsupervised Speech Recognition Without Pronunciation Models","date":"2024-06-12","arxiv_id":"2406.08380","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-unsupervised-speech-recognition#ran","syntology_url":"https://syntology.ai/paper/2406.08380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08380"}},"official":{"repos":["jeromeni/wholeword-uasr-jstti"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/less-peaky-and-more-accurate-ctc-forced","slug":"less-peaky-and-more-accurate-ctc-forced","title":"Less Peaky and More Accurate CTC Forced Alignment by Label Priors","date":"2024-04-22","arxiv_id":"2406.02560","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/less-peaky-and-more-accurate-ctc-forced#ran","syntology_url":"https://syntology.ai/paper/2406.02560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.02560"}},"official":{"repos":["huangruizhe/audio"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/asr-advancements-for-indigenous-languages","slug":"asr-advancements-for-indigenous-languages","title":"Automatic Speech Recognition Advancements for Indigenous Languages of the Americas","date":"2024-04-12","arxiv_id":"2404.08368","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/asr-advancements-for-indigenous-languages#ran","syntology_url":"https://syntology.ai/paper/2404.08368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08368"}},"official":null}},{"url":"/paper/speech-robust-bench-a-robustness-benchmark","slug":"speech-robust-bench-a-robustness-benchmark","title":"Speech Robust Bench: A Robustness Benchmark For Speech Recognition","date":"2024-03-08","arxiv_id":"2403.07937","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/speech-robust-bench-a-robustness-benchmark#ran","syntology_url":"https://syntology.ai/paper/2403.07937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07937"}},"official":{"repos":["ahmedshah1494/speech_robust_bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/how-do-hyenas-deal-with-human-speech-speech","slug":"how-do-hyenas-deal-with-human-speech-speech","title":"How do Hyenas deal with Human Speech? Speech Recognition and Translation with ConfHyena","date":"2024-02-20","arxiv_id":"2402.13208","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-do-hyenas-deal-with-human-speech-speech#ran","syntology_url":"https://syntology.ai/paper/2402.13208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13208"}},"official":{"repos":["hlt-mt/fbk-fairseq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/it-s-never-too-late-fusing-acoustic","slug":"it-s-never-too-late-fusing-acoustic","title":"It's Never Too Late: Fusing Acoustic Information into Large Language Models for Automatic Speech Recognition","date":"2024-02-08","arxiv_id":"2402.05457","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/it-s-never-too-late-fusing-acoustic#ran","syntology_url":"https://syntology.ai/paper/2402.05457","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05457"}},"official":null}},{"url":"/paper/unified-speech-text-pretraining-for-spoken","slug":"unified-speech-text-pretraining-for-spoken","title":"Paralinguistics-Aware Speech-Empowered Large Language Models for Natural Conversation","date":"2024-02-08","arxiv_id":"2402.05706","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":5,"n_ran_checked":7,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unified-speech-text-pretraining-for-spoken#ran","syntology_url":"https://syntology.ai/paper/2402.05706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05706"}},"official":{"repos":["naver-ai/usdm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/reborn-reinforcement-learned-boundary","slug":"reborn-reinforcement-learned-boundary","title":"REBORN: Reinforcement-Learned Boundary Segmentation with Iterative Training for Unsupervised ASR","date":"2024-02-06","arxiv_id":"2402.03988","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/reborn-reinforcement-learned-boundary#ran","syntology_url":"https://syntology.ai/paper/2402.03988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03988"}},"official":{"repos":["andybi7676/reborn-uasr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-efficient-learners","slug":"large-language-models-are-efficient-learners","title":"Large Language Models are Efficient Learners of Noise-Robust Speech Recognition","date":"2024-01-19","arxiv_id":"2401.10446","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-are-efficient-learners#ran","syntology_url":"https://syntology.ai/paper/2401.10446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10446"}},"official":{"repos":["yuchen005/robustger"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multichannel-av-wav2vec2-a-framework-for","slug":"multichannel-av-wav2vec2-a-framework-for","title":"Multichannel AV-wav2vec2: A Framework for Learning Multichannel Multi-Modal Speech Representation","date":"2024-01-07","arxiv_id":"2401.03468","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":2,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"8 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multichannel-av-wav2vec2-a-framework-for#ran","syntology_url":"https://syntology.ai/paper/2401.03468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03468"}},"official":{"repos":["zqs01/multi-channel-wav2vec2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diarizationlm-speaker-diarization-post","slug":"diarizationlm-speaker-diarization-post","title":"DiarizationLM: Speaker Diarization Post-Processing with Large Language Models","date":"2024-01-07","arxiv_id":"2401.03506","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diarizationlm-speaker-diarization-post#ran","syntology_url":"https://syntology.ai/paper/2401.03506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03506"}},"official":{"repos":["google/speaker-id"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/advancing-test-time-adaptation-for-acoustic","slug":"advancing-test-time-adaptation-for-acoustic","title":"Advancing Test-Time Adaptation in Wild Acoustic Test Settings","date":"2023-10-14","arxiv_id":"2310.09505","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/advancing-test-time-adaptation-for-acoustic#ran","syntology_url":"https://syntology.ai/paper/2310.09505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09505"}},"official":{"repos":["Waffle-Liu/CEA"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/whispering-llama-a-cross-modal-generative","slug":"whispering-llama-a-cross-modal-generative","title":"Whispering LLaMA: A Cross-Modal Generative Error Correction Framework for Speech Recognition","date":"2023-10-10","arxiv_id":"2310.06434","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/whispering-llama-a-cross-modal-generative#ran","syntology_url":"https://syntology.ai/paper/2310.06434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06434"}},"official":{"repos":["srijith-rkr/whispering-llama"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lauragpt-listen-attend-understand-and","slug":"lauragpt-listen-attend-understand-and","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","date":"2023-10-07","arxiv_id":"2310.04673","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lauragpt-listen-attend-understand-and#ran","syntology_url":"https://syntology.ai/paper/2310.04673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04673"}},"official":null}},{"url":"/paper/howtocaption-prompting-llms-to-transform","slug":"howtocaption-prompting-llms-to-transform","title":"HowToCaption: Prompting LLMs to Transform Video Annotations at Scale","date":"2023-10-07","arxiv_id":"2310.04900","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/howtocaption-prompting-llms-to-transform#ran","syntology_url":"https://syntology.ai/paper/2310.04900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04900"}},"official":{"repos":["ninatu/howtocaption"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/librispeech-pc-benchmark-for-evaluation-of","slug":"librispeech-pc-benchmark-for-evaluation-of","title":"LibriSpeech-PC: Benchmark for Evaluation of Punctuation and Capitalization Capabilities of end-to-end ASR Models","date":"2023-10-04","arxiv_id":"2310.02943","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/librispeech-pc-benchmark-for-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2310.02943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02943"}},"official":null}},{"url":"/paper/federated-learning-with-differential-privacy","slug":"federated-learning-with-differential-privacy","title":"Enabling Differentially Private Federated Learning for Speech Recognition: Benchmarks, Adaptive Optimizers and Gradient Clipping","date":"2023-09-29","arxiv_id":"2310.00098","repositories_listed":0,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/federated-learning-with-differential-privacy#ran","syntology_url":"https://syntology.ai/paper/2310.00098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00098"}},"official":null}},{"url":"/paper/hyporadise-an-open-baseline-for-generative-1","slug":"hyporadise-an-open-baseline-for-generative-1","title":"HyPoradise: An Open Baseline for Generative Speech Recognition with Large Language Models","date":"2023-09-27","arxiv_id":"2309.15701","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hyporadise-an-open-baseline-for-generative-1#ran","syntology_url":"https://syntology.ai/paper/2309.15701","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15701"}},"official":{"repos":["hypotheses-paradise/hypo2trans"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/encodecmae-leveraging-neural-codecs-for","slug":"encodecmae-leveraging-neural-codecs-for","title":"EnCodecMAE: Leveraging neural codecs for universal audio representation learning","date":"2023-09-14","arxiv_id":"2309.07391","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/encodecmae-leveraging-neural-codecs-for#ran","syntology_url":"https://syntology.ai/paper/2309.07391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07391"}},"official":{"repos":["habla-liaa/encodecmae"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seamlessm4t-massively-multilingual-multimodal","slug":"seamlessm4t-massively-multilingual-multimodal","title":"SeamlessM4T: Massively Multilingual & Multimodal Machine Translation","date":"2023-08-22","arxiv_id":"2308.11596","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seamlessm4t-massively-multilingual-multimodal#ran","syntology_url":"https://syntology.ai/paper/2308.11596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11596"}},"official":{"repos":["facebookresearch/seamless_communication"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quilt-1m-one-million-image-text-pairs-for-1","slug":"quilt-1m-one-million-image-text-pairs-for-1","title":"Quilt-1M: One Million Image-Text Pairs for Histopathology","date":"2023-06-20","arxiv_id":"2306.11207","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quilt-1m-one-million-image-text-pairs-for-1#ran","syntology_url":"https://syntology.ai/paper/2306.11207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11207"}},"official":{"repos":["wisdomikezogwo/quilt1m"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sgem-test-time-adaptation-for-automatic","slug":"sgem-test-time-adaptation-for-automatic","title":"SGEM: Test-Time Adaptation for Automatic Speech Recognition via Sequential-Level Generalized Entropy Minimization","date":"2023-06-03","arxiv_id":"2306.01981","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sgem-test-time-adaptation-for-automatic#ran","syntology_url":"https://syntology.ai/paper/2306.01981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01981"}},"official":{"repos":["drumpt/sgem"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/improved-deepfake-detection-using-whisper","slug":"improved-deepfake-detection-using-whisper","title":"Improved DeepFake Detection Using Whisper Features","date":"2023-06-02","arxiv_id":"2306.01428","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improved-deepfake-detection-using-whisper#ran","syntology_url":"https://syntology.ai/paper/2306.01428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01428"}},"official":{"repos":["piotrkawa/deepfake-whisper-features"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-speech-technology-to-1000-languages-1","slug":"scaling-speech-technology-to-1000-languages-1","title":"Scaling Speech Technology to 1,000+ Languages","date":"2023-05-22","arxiv_id":"2305.13516","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-speech-technology-to-1000-languages-1#ran","syntology_url":"https://syntology.ai/paper/2305.13516","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13516"}},"official":{"repos":["facebookresearch/fairseq","pytorch/fairseq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/making-more-of-little-data-improving-low","slug":"making-more-of-little-data-improving-low","title":"Making More of Little Data: Improving Low-Resource Automatic Speech Recognition Using Data Augmentation","date":"2023-05-18","arxiv_id":"2305.10951","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-more-of-little-data-improving-low#ran","syntology_url":"https://syntology.ai/paper/2305.10951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10951"}},"official":{"repos":["bartelds/asr-augmentation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/auto-avsr-audio-visual-speech-recognition","slug":"auto-avsr-audio-visual-speech-recognition","title":"Auto-AVSR: Audio-Visual Speech Recognition with Automatic Labels","date":"2023-03-25","arxiv_id":"2303.14307","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/auto-avsr-audio-visual-speech-recognition#ran","syntology_url":"https://syntology.ai/paper/2303.14307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14307"}},"official":{"repos":["mpc001/auto_avsr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/stabilizing-transformer-training-by","slug":"stabilizing-transformer-training-by","title":"Stabilizing Transformer Training by Preventing Attention Entropy Collapse","date":"2023-03-11","arxiv_id":"2303.06296","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stabilizing-transformer-training-by#ran","syntology_url":"https://syntology.ai/paper/2303.06296","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06296"}},"official":{"repos":["apple/ml-sigma-reparam"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/complex-dynamic-neurons-improved-spiking","slug":"complex-dynamic-neurons-improved-spiking","title":"Complex Dynamic Neurons Improved Spiking Transformer Network for Efficient Automatic Speech Recognition","date":"2023-02-02","arxiv_id":"2302.01194","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/complex-dynamic-neurons-improved-spiking#ran","syntology_url":"https://syntology.ai/paper/2302.01194","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.01194"}},"official":{"repos":["MingLunHan/CIF-PyTorch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/audio-visual-efficient-conformer-for-robust","slug":"audio-visual-efficient-conformer-for-robust","title":"Audio-Visual Efficient Conformer for Robust Speech Recognition","date":"2023-01-04","arxiv_id":"2301.01456","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/audio-visual-efficient-conformer-for-robust#ran","syntology_url":"https://syntology.ai/paper/2301.01456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.01456"}},"official":{"repos":["burchim/avec"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-voice-reconstruction-from-eeg-during","slug":"towards-voice-reconstruction-from-eeg-during","title":"Towards Voice Reconstruction from EEG during Imagined Speech","date":"2023-01-02","arxiv_id":"2301.07173","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-voice-reconstruction-from-eeg-during#ran","syntology_url":"https://syntology.ai/paper/2301.07173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07173"}},"official":{"repos":["youngeun1209/neurotalk"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/softcorrect-error-correction-with-soft","slug":"softcorrect-error-correction-with-soft","title":"SoftCorrect: Error Correction with Soft Detection for Automatic Speech Recognition","date":"2022-12-02","arxiv_id":"2212.01039","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/softcorrect-error-correction-with-soft#ran","syntology_url":"https://syntology.ai/paper/2212.01039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.01039"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/euro-espnet-unsupervised-asr-open-source","slug":"euro-espnet-unsupervised-asr-open-source","title":"EURO: ESPnet Unsupervised ASR Open-source Toolkit","date":"2022-11-30","arxiv_id":"2211.17196","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/euro-espnet-unsupervised-asr-open-source#ran","syntology_url":"https://syntology.ai/paper/2211.17196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.17196"}},"official":{"repos":["espnet/espnet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/esb-a-benchmark-for-multi-domain-end-to-end","slug":"esb-a-benchmark-for-multi-domain-end-to-end","title":"ESB: A Benchmark For Multi-Domain End-to-End Speech Recognition","date":"2022-10-24","arxiv_id":"2210.13352","repositories_listed":2,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/esb-a-benchmark-for-multi-domain-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2210.13352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13352"}},"official":null}},{"url":"/paper/can-we-use-common-voice-to-train-a-multi","slug":"can-we-use-common-voice-to-train-a-multi","title":"Can we use Common Voice to train a Multi-Speaker TTS system?","date":"2022-10-12","arxiv_id":"2210.06370","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-we-use-common-voice-to-train-a-multi#ran","syntology_url":"https://syntology.ai/paper/2210.06370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06370"}},"official":null}},{"url":"/paper/joeys2t-minimalistic-speech-to-text-modeling","slug":"joeys2t-minimalistic-speech-to-text-modeling","title":"JoeyS2T: Minimalistic Speech-to-Text Modeling with JoeyNMT","date":"2022-10-05","arxiv_id":"2210.02545","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/joeys2t-minimalistic-speech-to-text-modeling#ran","syntology_url":"https://syntology.ai/paper/2210.02545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.02545"}},"official":{"repos":["may-/joeys2t"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/indicsuperb-a-speech-processing-universal","slug":"indicsuperb-a-speech-processing-universal","title":"IndicSUPERB: A Speech Processing Universal Performance Benchmark for Indian languages","date":"2022-08-24","arxiv_id":"2208.11761","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/indicsuperb-a-speech-processing-universal#ran","syntology_url":"https://syntology.ai/paper/2208.11761","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.11761"}},"official":{"repos":["AI4Bharat/indicSUPERB"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-transfer-learning-of-wav2vec-2-0-for","slug":"towards-transfer-learning-of-wav2vec-2-0-for","title":"Transfer Learning of wav2vec 2.0 for Automatic Lyric Transcription","date":"2022-07-20","arxiv_id":"2207.09747","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-transfer-learning-of-wav2vec-2-0-for#ran","syntology_url":"https://syntology.ai/paper/2207.09747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.09747"}},"official":{"repos":["guxm2021/alt_speechbrain"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/squeezeformer-an-efficient-transformer-for","slug":"squeezeformer-an-efficient-transformer-for","title":"Squeezeformer: An Efficient Transformer for Automatic Speech Recognition","date":"2022-06-02","arxiv_id":"2206.00888","repositories_listed":4,"syntology":{"n":49,"n_ran":31,"n_constructed":12,"n_ran_checked":28,"n_instrument":3,"n_unverified":18,"n_honours":3,"n_violates":0,"n_no_contract":25,"n_pointer_only":0,"phrase":"31 ran (of which 12 constructed an object rather than computing a result; 28 with no instrument failure: 3 honoured, 0 violated, 25 with no contract checked; 3 where Syntology's instrument failed) · 18 unverified","sample_list":"/paper/squeezeformer-an-efficient-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2206.00888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.00888"}},"official":{"repos":["kssteven418/squeezeformer"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":10,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/language-models-with-image-descriptors-are","slug":"language-models-with-image-descriptors-are","title":"Language Models with Image Descriptors are Strong Few-Shot Video-Language Learners","date":"2022-05-22","arxiv_id":"2205.10747","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-with-image-descriptors-are#ran","syntology_url":"https://syntology.ai/paper/2205.10747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10747"}},"official":{"repos":["mikewangwzhl/vidil"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-scale-streaming-end-to-end-speech","slug":"large-scale-streaming-end-to-end-speech","title":"Large-Scale Streaming End-to-End Speech Translation with Neural Transducers","date":"2022-04-11","arxiv_id":"2204.05352","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-scale-streaming-end-to-end-speech#ran","syntology_url":"https://syntology.ai/paper/2204.05352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.05352"}},"official":null}},{"url":"/paper/unsupervised-uncertainty-measures-of","slug":"unsupervised-uncertainty-measures-of","title":"Unsupervised Uncertainty Measures of Automatic Speech Recognition for Non-intrusive Speech Intelligibility Prediction","date":"2022-04-08","arxiv_id":"2204.04288","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsupervised-uncertainty-measures-of#ran","syntology_url":"https://syntology.ai/paper/2204.04288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.04288"}},"official":{"repos":["claritychallenge/clarity"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lighthubert-lightweight-and-configurable","slug":"lighthubert-lightweight-and-configurable","title":"LightHuBERT: Lightweight and Configurable Speech Representation Learning with Once-for-All Hidden-Unit BERT","date":"2022-03-29","arxiv_id":"2203.15610","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/lighthubert-lightweight-and-configurable#ran","syntology_url":"https://syntology.ai/paper/2203.15610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15610"}},"official":{"repos":["mechanicalsea/lighthubert"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/cmgan-conformer-based-metric-gan-for-speech","slug":"cmgan-conformer-based-metric-gan-for-speech","title":"CMGAN: Conformer-based Metric GAN for Speech Enhancement","date":"2022-03-28","arxiv_id":"2203.15149","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":2,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 3 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cmgan-conformer-based-metric-gan-for-speech#ran","syntology_url":"https://syntology.ai/paper/2203.15149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15149"}},"official":{"repos":["ruizhecao96/cmgan"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-adapter-transfer-of-self-supervised","slug":"efficient-adapter-transfer-of-self-supervised","title":"Efficient Adapter Transfer of Self-Supervised Speech Models for Automatic Speech Recognition","date":"2022-02-07","arxiv_id":"2202.03218","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/efficient-adapter-transfer-of-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2202.03218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03218"}},"official":null}},{"url":"/paper/robust-self-supervised-audio-visual-speech","slug":"robust-self-supervised-audio-visual-speech","title":"Robust Self-Supervised Audio-Visual Speech Recognition","date":"2022-01-05","arxiv_id":"2201.01763","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-self-supervised-audio-visual-speech#ran","syntology_url":"https://syntology.ai/paper/2201.01763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.01763"}},"official":{"repos":["facebookresearch/av_hubert"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/slue-new-benchmark-tasks-for-spoken-language","slug":"slue-new-benchmark-tasks-for-spoken-language","title":"SLUE: New Benchmark Tasks for Spoken Language Understanding Evaluation on Natural Speech","date":"2021-11-19","arxiv_id":"2111.10367","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/slue-new-benchmark-tasks-for-spoken-language#ran","syntology_url":"https://syntology.ai/paper/2111.10367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.10367"}},"official":{"repos":["asappresearch/slue-toolkit"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/speecht5-unified-modal-encoder-decoder-pre","slug":"speecht5-unified-modal-encoder-decoder-pre","title":"SpeechT5: Unified-Modal Encoder-Decoder Pre-Training for Spoken Language Processing","date":"2021-10-14","arxiv_id":"2110.07205","repositories_listed":6,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/speecht5-unified-modal-encoder-decoder-pre#ran","syntology_url":"https://syntology.ai/paper/2110.07205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.07205"}},"official":{"repos":["microsoft/speecht5"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/fastcorrect-2-fast-error-correction-on","slug":"fastcorrect-2-fast-error-correction-on","title":"FastCorrect 2: Fast Error Correction on Multiple Candidates for Automatic Speech Recognition","date":"2021-09-29","arxiv_id":"2109.14420","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fastcorrect-2-fast-error-correction-on#ran","syntology_url":"https://syntology.ai/paper/2109.14420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.14420"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/strode-stochastic-boundary-ordinary","slug":"strode-stochastic-boundary-ordinary","title":"STRODE: Stochastic Boundary Ordinary Differential Equation","date":"2021-07-17","arxiv_id":"2107.08273","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/strode-stochastic-boundary-ordinary#ran","syntology_url":"https://syntology.ai/paper/2107.08273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.08273"}},"official":{"repos":["Waffle-Liu/STRODE"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/pretext-tasks-selection-for-multitask-self","slug":"pretext-tasks-selection-for-multitask-self","title":"Pretext Tasks selection for multitask self-supervised speech representation learning","date":"2021-07-01","arxiv_id":"2107.00594","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pretext-tasks-selection-for-multitask-self#ran","syntology_url":"https://syntology.ai/paper/2107.00594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.00594"}},"official":{"repos":["salah-zaiem/PL-groupselection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/attention-based-contextual-language-model","slug":"attention-based-contextual-language-model","title":"Attention-based Contextual Language Model Adaptation for Speech Recognition","date":"2021-06-02","arxiv_id":"2106.01451","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/attention-based-contextual-language-model#ran","syntology_url":"https://syntology.ai/paper/2106.01451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.01451"}},"official":{"repos":["amazon-research/contextual-attention-nlm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/waveguard-understanding-and-mitigating-audio","slug":"waveguard-understanding-and-mitigating-audio","title":"WaveGuard: Understanding and Mitigating Audio Adversarial Examples","date":"2021-03-04","arxiv_id":"2103.03344","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/waveguard-understanding-and-mitigating-audio#ran","syntology_url":"https://syntology.ai/paper/2103.03344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.03344"}},"official":null}},{"url":"/paper/decentralizing-feature-extraction-with","slug":"decentralizing-feature-extraction-with","title":"Decentralizing Feature Extraction with Quantum Convolutional Neural Network for Automatic Speech Recognition","date":"2020-10-26","arxiv_id":"2010.13309","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/decentralizing-feature-extraction-with#ran","syntology_url":"https://syntology.ai/paper/2010.13309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.13309"}},"official":{"repos":["huckiyang/speech_quantum_dl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/pushing-the-limits-of-semi-supervised","slug":"pushing-the-limits-of-semi-supervised","title":"Pushing the Limits of Semi-Supervised Learning for Automatic Speech Recognition","date":"2020-10-20","arxiv_id":"2010.10504","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pushing-the-limits-of-semi-supervised#ran","syntology_url":"https://syntology.ai/paper/2010.10504","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10504"}},"official":null}},{"url":"/paper/representation-learning-for-sequence-data-1","slug":"representation-learning-for-sequence-data-1","title":"Representation Learning for Sequence Data with Deep Autoencoding Predictive Components","date":"2020-10-07","arxiv_id":"2010.03135","repositories_listed":2,"syntology":{"n":29,"n_ran":22,"n_constructed":0,"n_ran_checked":22,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":22,"n_pointer_only":2,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 22 with no instrument failure: 0 honoured, 0 violated, 22 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/representation-learning-for-sequence-data-1#ran","syntology_url":"https://syntology.ai/paper/2010.03135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.03135"}},"official":{"repos":["JunwenBai/DAPC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed"]}}},{"url":"/paper/on-the-comparison-of-popular-end-to-end","slug":"on-the-comparison-of-popular-end-to-end","title":"On the Comparison of Popular End-to-End Models for Large Scale Speech Recognition","date":"2020-05-28","arxiv_id":"2005.14327","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-the-comparison-of-popular-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2005.14327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.14327"}},"official":{"repos":["cywang97/StreamingTransformer"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-monotonic-multihead-attention-for","slug":"enhancing-monotonic-multihead-attention-for","title":"Enhancing Monotonic Multihead Attention for Streaming ASR","date":"2020-05-19","arxiv_id":"2005.09394","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/enhancing-monotonic-multihead-attention-for#ran","syntology_url":"https://syntology.ai/paper/2005.09394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.09394"}},"official":{"repos":["hirofumi0810/neural_sp"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/conformer-convolution-augmented-transformer","slug":"conformer-convolution-augmented-transformer","title":"Conformer: Convolution-augmented Transformer for Speech Recognition","date":"2020-05-16","arxiv_id":"2005.08100","repositories_listed":25,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":3,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 3 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conformer-convolution-augmented-transformer#ran","syntology_url":"https://syntology.ai/paper/2005.08100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08100"}},"official":null}},{"url":"/paper/unsupervised-pretraining-transfers-well","slug":"unsupervised-pretraining-transfers-well","title":"Unsupervised pretraining transfers well across languages","date":"2020-02-07","arxiv_id":"2002.02848","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-pretraining-transfers-well#ran","syntology_url":"https://syntology.ai/paper/2002.02848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.02848"}},"official":{"repos":["facebookresearch/CPC_audio"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/espnet-tts-unified-reproducible-and","slug":"espnet-tts-unified-reproducible-and","title":"ESPnet-TTS: Unified, Reproducible, and Integratable Open Source End-to-End Text-to-Speech Toolkit","date":"2019-10-24","arxiv_id":"1910.10909","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/espnet-tts-unified-reproducible-and#ran","syntology_url":"https://syntology.ai/paper/1910.10909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.10909"}},"official":{"repos":["r9y9/wavenet_vocoder","espnet/espnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/multilingual-end-to-end-speech-translation","slug":"multilingual-end-to-end-speech-translation","title":"Multilingual End-to-End Speech Translation","date":"2019-10-01","arxiv_id":"1910.00254","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multilingual-end-to-end-speech-translation#ran","syntology_url":"https://syntology.ai/paper/1910.00254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.00254"}},"official":{"repos":["espnet/espnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/espresso-a-fast-end-to-end-neural-speech","slug":"espresso-a-fast-end-to-end-neural-speech","title":"Espresso: A Fast End-to-end Neural Speech Recognition Toolkit","date":"2019-09-18","arxiv_id":"1909.08723","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/espresso-a-fast-end-to-end-neural-speech#ran","syntology_url":"https://syntology.ai/paper/1909.08723","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.08723"}},"official":{"repos":["freewym/espresso"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/specaugment-a-simple-data-augmentation-method","slug":"specaugment-a-simple-data-augmentation-method","title":"SpecAugment: A Simple Data Augmentation Method for Automatic Speech Recognition","date":"2019-04-18","arxiv_id":"1904.08779","repositories_listed":30,"syntology":{"n":18,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/specaugment-a-simple-data-augmentation-method#ran","syntology_url":"https://syntology.ai/paper/1904.08779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.08779"}},"official":null}},{"url":"/paper/snips-voice-platform-an-embedded-spoken","slug":"snips-voice-platform-an-embedded-spoken","title":"Snips Voice Platform: an embedded Spoken Language Understanding system for private-by-design voice interfaces","date":"2018-05-25","arxiv_id":"1805.10190","repositories_listed":16,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/snips-voice-platform-an-embedded-spoken#ran","syntology_url":"https://syntology.ai/paper/1805.10190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.10190"}},"official":{"repos":["snipsco/nlu-benchmark","snipsco/snips-nlu"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/targeted-adversarial-examples-for-black-box","slug":"targeted-adversarial-examples-for-black-box","title":"Targeted Adversarial Examples for Black Box Audio Systems","date":"2018-05-20","arxiv_id":"1805.07820","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/targeted-adversarial-examples-for-black-box#ran","syntology_url":"https://syntology.ai/paper/1805.07820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.07820"}},"official":{"repos":["rtaori/Black-Box-Audio"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/speech-commands-a-dataset-for-limited","slug":"speech-commands-a-dataset-for-limited","title":"Speech Commands: A Dataset for Limited-Vocabulary Speech Recognition","date":"2018-04-09","arxiv_id":"1804.03209","repositories_listed":35,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/speech-commands-a-dataset-for-limited#ran","syntology_url":"https://syntology.ai/paper/1804.03209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.03209"}},"official":null}},{"url":"/paper/a-baseline-for-detecting-misclassified-and","slug":"a-baseline-for-detecting-misclassified-and","title":"A Baseline for Detecting Misclassified and Out-of-Distribution Examples in Neural Networks","date":"2016-10-07","arxiv_id":"1610.02136","repositories_listed":14,"syntology":{"n":21,"n_ran":19,"n_constructed":0,"n_ran_checked":16,"n_instrument":3,"n_unverified":2,"n_honours":4,"n_violates":0,"n_no_contract":12,"n_pointer_only":6,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 4 honoured, 0 violated, 12 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-baseline-for-detecting-misclassified-and#ran","syntology_url":"https://syntology.ai/paper/1610.02136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1610.02136"}},"official":{"repos":["hendrycks/error-detection"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/neural-nilm-deep-neural-networks-applied-to","slug":"neural-nilm-deep-neural-networks-applied-to","title":"Neural NILM: Deep Neural Networks Applied to Energy Disaggregation","date":"2015-07-23","arxiv_id":"1507.06594","repositories_listed":4,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/neural-nilm-deep-neural-networks-applied-to#ran","syntology_url":"https://syntology.ai/paper/1507.06594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1507.06594"}},"official":{"repos":["JackKelly/neuralnilm_prototype"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["listed","official"]}}}],"record_sha256":"b5cc79ac343e74512fab8b8146eeb61148ba920730e10c96f99d9d6960aebb79","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}