{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/ran/1","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":162,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1/papers/ran/1","prev":null,"next":"/task/speech-recognition-1/papers/ran/2","papers":[{"url":"/paper/unifying-streaming-and-non-streaming","slug":"unifying-streaming-and-non-streaming","title":"Unifying Streaming and Non-streaming Zipformer-based ASR","date":"2025-06-17","arxiv_id":"2506.14434","repositories_listed":0,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unifying-streaming-and-non-streaming#ran","syntology_url":"https://syntology.ai/paper/2506.14434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.14434"}},"official":null}},{"url":"/paper/cosyvoice-3-towards-in-the-wild-speech","slug":"cosyvoice-3-towards-in-the-wild-speech","title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","date":"2025-05-23","arxiv_id":"2505.17589","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cosyvoice-3-towards-in-the-wild-speech#ran","syntology_url":"https://syntology.ai/paper/2505.17589","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17589"}},"official":{"repos":["funaudiollm/cosyvoice"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/daily-omni-towards-audio-visual-reasoning","slug":"daily-omni-towards-audio-visual-reasoning","title":"Daily-Omni: Towards Audio-Visual Reasoning with Temporal Alignment across Modalities","date":"2025-05-23","arxiv_id":"2505.17862","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/daily-omni-towards-audio-visual-reasoning#ran","syntology_url":"https://syntology.ai/paper/2505.17862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17862"}},"official":{"repos":["lliar-liar/daily-omni"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/multi-head-temporal-latent-attention","slug":"multi-head-temporal-latent-attention","title":"Multi-head Temporal Latent Attention","date":"2025-05-19","arxiv_id":"2505.13544","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-head-temporal-latent-attention#ran","syntology_url":"https://syntology.ai/paper/2505.13544","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13544"}},"official":{"repos":["d-keqi/mlta"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/voila-voice-language-foundation-models-for","slug":"voila-voice-language-foundation-models-for","title":"Voila: Voice-Language Foundation Models for Real-Time Autonomous Interaction and Voice Role-Play","date":"2025-05-05","arxiv_id":"2505.02707","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/voila-voice-language-foundation-models-for#ran","syntology_url":"https://syntology.ai/paper/2505.02707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02707"}},"official":{"repos":["maitrix-org/voila"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":6,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/kimi-audio-technical-report","slug":"kimi-audio-technical-report","title":"Kimi-Audio Technical Report","date":"2025-04-25","arxiv_id":"2504.18425","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/kimi-audio-technical-report#ran","syntology_url":"https://syntology.ai/paper/2504.18425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.18425"}},"official":{"repos":["moonshotai/kimi-audio"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/whispering-under-the-eaves-protecting-user","slug":"whispering-under-the-eaves-protecting-user","title":"Whispering Under the Eaves: Protecting User Privacy Against Commercial and LLM-powered Automatic Speech Recognition Systems","date":"2025-04-01","arxiv_id":"2504.00858","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/whispering-under-the-eaves-protecting-user#ran","syntology_url":"https://syntology.ai/paper/2504.00858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.00858"}},"official":{"repos":["WeifeiJin/AudioShield"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/liteasr-efficient-automatic-speech","slug":"liteasr-efficient-automatic-speech","title":"LiteASR: Efficient Automatic Speech Recognition with Low-Rank Approximation","date":"2025-02-27","arxiv_id":"2502.20583","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/liteasr-efficient-automatic-speech#ran","syntology_url":"https://syntology.ai/paper/2502.20583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20583"}},"official":{"repos":["efeslab/liteasr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dypcl-dynamic-phoneme-level-contrastive","slug":"dypcl-dynamic-phoneme-level-contrastive","title":"DyPCL: Dynamic Phoneme-level Contrastive Learning for Dysarthric Speech Recognition","date":"2025-01-31","arxiv_id":"2501.19010","repositories_listed":0,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dypcl-dynamic-phoneme-level-contrastive#ran","syntology_url":"https://syntology.ai/paper/2501.19010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19010"}},"official":null}},{"url":"/paper/multi-task-corrupted-prediction-for-learning","slug":"multi-task-corrupted-prediction-for-learning","title":"Multi-Task Corrupted Prediction for Learning Robust Audio-Visual Speech Representation","date":"2025-01-23","arxiv_id":"2504.18539","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-corrupted-prediction-for-learning#ran","syntology_url":"https://syntology.ai/paper/2504.18539","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.18539"}},"official":{"repos":["sungnyun/cav2vec"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/glm-4-voice-towards-intelligent-and-human","slug":"glm-4-voice-towards-intelligent-and-human","title":"GLM-4-Voice: Towards Intelligent and Human-Like End-to-End Spoken Chatbot","date":"2024-12-03","arxiv_id":"2412.02612","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/glm-4-voice-towards-intelligent-and-human#ran","syntology_url":"https://syntology.ai/paper/2412.02612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02612"}},"official":{"repos":["thudm/glm-4-voice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unified-speech-recognition-a-single-model-for","slug":"unified-speech-recognition-a-single-model-for","title":"Unified Speech Recognition: A Single Model for Auditory, Visual, and Audiovisual Inputs","date":"2024-11-04","arxiv_id":"2411.02256","repositories_listed":1,"syntology":{"n":9,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":9,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/unified-speech-recognition-a-single-model-for#ran","syntology_url":"https://syntology.ai/paper/2411.02256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02256"}},"official":{"repos":["ahaliassos/usr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/emg2qwerty-a-large-dataset-with-baselines-for","slug":"emg2qwerty-a-large-dataset-with-baselines-for","title":"emg2qwerty: A Large Dataset with Baselines for Touch Typing using Surface Electromyography","date":"2024-10-26","arxiv_id":"2410.20081","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emg2qwerty-a-large-dataset-with-baselines-for#ran","syntology_url":"https://syntology.ai/paper/2410.20081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20081"}},"official":{"repos":["facebookresearch/emg2qwerty"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/voicebench-benchmarking-llm-based-voice","slug":"voicebench-benchmarking-llm-based-voice","title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","date":"2024-10-22","arxiv_id":"2410.17196","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/voicebench-benchmarking-llm-based-voice#ran","syntology_url":"https://syntology.ai/paper/2410.17196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17196"}},"official":{"repos":["matthewcym/voicebench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/moonshine-speech-recognition-for-live","slug":"moonshine-speech-recognition-for-live","title":"Moonshine: Speech Recognition for Live Transcription and Voice Commands","date":"2024-10-21","arxiv_id":"2410.15608","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/moonshine-speech-recognition-for-live#ran","syntology_url":"https://syntology.ai/paper/2410.15608","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15608"}},"official":{"repos":["usefulsensors/moonshine"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/eh-mam-easy-to-hard-masked-acoustic-modeling","slug":"eh-mam-easy-to-hard-masked-acoustic-modeling","title":"EH-MAM: Easy-to-Hard Masked Acoustic Modeling for Self-Supervised Speech Representation Learning","date":"2024-10-17","arxiv_id":"2410.13179","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/eh-mam-easy-to-hard-masked-acoustic-modeling#ran","syntology_url":"https://syntology.ai/paper/2410.13179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13179"}},"official":{"repos":["cs20s030/ehmam"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-strong-audio-visual","slug":"large-language-models-are-strong-audio-visual","title":"Large Language Models are Strong Audio-Visual Speech Recognition Learners","date":"2024-09-18","arxiv_id":"2409.12319","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":2,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"9 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/large-language-models-are-strong-audio-visual#ran","syntology_url":"https://syntology.ai/paper/2409.12319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12319"}},"official":{"repos":["umbertocappellazzo/llama-avsr"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/moshi-a-speech-text-foundation-model-for-real","slug":"moshi-a-speech-text-foundation-model-for-real","title":"Moshi: a speech-text foundation model for real-time dialogue","date":"2024-09-17","arxiv_id":"2410.00037","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/moshi-a-speech-text-foundation-model-for-real#ran","syntology_url":"https://syntology.ai/paper/2410.00037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00037"}},"official":{"repos":["kyutai-labs/moshi"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/approaching-deep-learning-through-the","slug":"approaching-deep-learning-through-the","title":"Approaching Deep Learning through the Spectral Dynamics of Weights","date":"2024-08-21","arxiv_id":"2408.11804","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/approaching-deep-learning-through-the#ran","syntology_url":"https://syntology.ai/paper/2408.11804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11804"}},"official":{"repos":["dyunis/spectral_dynamics"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/the-llama-3-herd-of-models","slug":"the-llama-3-herd-of-models","title":"The Llama 3 Herd of Models","date":"2024-07-31","arxiv_id":"2407.21783","repositories_listed":5,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-llama-3-herd-of-models#ran","syntology_url":"https://syntology.ai/paper/2407.21783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21783"}},"official":null}},{"url":"/paper/dmel-speech-tokenization-made-simple","slug":"dmel-speech-tokenization-made-simple","title":"dMel: Speech Tokenization made Simple","date":"2024-07-22","arxiv_id":"2407.15835","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dmel-speech-tokenization-made-simple#ran","syntology_url":"https://syntology.ai/paper/2407.15835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15835"}},"official":{"repos":["apple/dmel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00005","slug":"2408-00005","title":"Framework for Curating Speech Datasets and Evaluating ASR Systems: A Case Study for Polish","date":"2024-07-18","arxiv_id":"2408.00005","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2408-00005#ran","syntology_url":"https://syntology.ai/paper/2408.00005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00005"}},"official":{"repos":["goodmike31/pl-asr-bigos-tools"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/learning-video-temporal-dynamics-with-cross","slug":"learning-video-temporal-dynamics-with-cross","title":"Learning Video Temporal Dynamics with Cross-Modal Attention for Robust Audio-Visual Speech Recognition","date":"2024-07-04","arxiv_id":"2407.03563","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":2,"n_ran_checked":8,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"12 ran (of which 2 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-video-temporal-dynamics-with-cross#ran","syntology_url":"https://syntology.ai/paper/2407.03563","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03563"}},"official":{"repos":["sungnyun/avsr-temporal-dynamics"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":2,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/funaudiollm-voice-understanding-and","slug":"funaudiollm-voice-understanding-and","title":"FunAudioLLM: Voice Understanding and Generation Foundation Models for Natural Interaction Between Humans and LLMs","date":"2024-07-04","arxiv_id":"2407.04051","repositories_listed":3,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/funaudiollm-voice-understanding-and#ran","syntology_url":"https://syntology.ai/paper/2407.04051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04051"}},"official":{"repos":["FunAudioLLM/SenseVoice","funaudiollm/cosyvoice"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/gigaspeech-2-an-evolving-large-scale-and","slug":"gigaspeech-2-an-evolving-large-scale-and","title":"GigaSpeech 2: An Evolving, Large-Scale and Multi-domain ASR Corpus for Low-Resource Languages with Automated Crawling, Transcription and Refinement","date":"2024-06-17","arxiv_id":"2406.11546","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gigaspeech-2-an-evolving-large-scale-and#ran","syntology_url":"https://syntology.ai/paper/2406.11546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11546"}},"official":{"repos":["SpeechColab/GigaSpeech2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/whisper-flamingo-integrating-visual-features","slug":"whisper-flamingo-integrating-visual-features","title":"Whisper-Flamingo: Integrating Visual Features into Whisper for Audio-Visual Speech Recognition and Translation","date":"2024-06-14","arxiv_id":"2406.10082","repositories_listed":1,"syntology":{"n":18,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":7,"n_honours":2,"n_violates":0,"n_no_contract":8,"n_pointer_only":18,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/whisper-flamingo-integrating-visual-features#ran","syntology_url":"https://syntology.ai/paper/2406.10082","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10082"}},"official":{"repos":["roudimit/whisper-flamingo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-unsupervised-speech-recognition","slug":"towards-unsupervised-speech-recognition","title":"Towards Unsupervised Speech Recognition Without Pronunciation Models","date":"2024-06-12","arxiv_id":"2406.08380","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-unsupervised-speech-recognition#ran","syntology_url":"https://syntology.ai/paper/2406.08380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08380"}},"official":{"repos":["jeromeni/wholeword-uasr-jstti"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/blsp-emo-towards-empathetic-large-speech","slug":"blsp-emo-towards-empathetic-large-speech","title":"BLSP-Emo: Towards Empathetic Large Speech-Language Models","date":"2024-06-06","arxiv_id":"2406.03872","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/blsp-emo-towards-empathetic-large-speech#ran","syntology_url":"https://syntology.ai/paper/2406.03872","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03872"}},"official":{"repos":["cwang621/blsp-emo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/streamspeech-simultaneous-speech-to-speech","slug":"streamspeech-simultaneous-speech-to-speech","title":"StreamSpeech: Simultaneous Speech-to-Speech Translation with Multi-task Learning","date":"2024-06-05","arxiv_id":"2406.03049","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streamspeech-simultaneous-speech-to-speech#ran","syntology_url":"https://syntology.ai/paper/2406.03049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03049"}},"official":{"repos":["ictnlp/streamspeech"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/less-peaky-and-more-accurate-ctc-forced","slug":"less-peaky-and-more-accurate-ctc-forced","title":"Less Peaky and More Accurate CTC Forced Alignment by Label Priors","date":"2024-04-22","arxiv_id":"2406.02560","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/less-peaky-and-more-accurate-ctc-forced#ran","syntology_url":"https://syntology.ai/paper/2406.02560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.02560"}},"official":{"repos":["huangruizhe/audio"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/asr-advancements-for-indigenous-languages","slug":"asr-advancements-for-indigenous-languages","title":"Automatic Speech Recognition Advancements for Indigenous Languages of the Americas","date":"2024-04-12","arxiv_id":"2404.08368","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/asr-advancements-for-indigenous-languages#ran","syntology_url":"https://syntology.ai/paper/2404.08368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08368"}},"official":null}},{"url":"/paper/braven-improving-self-supervised-pre-training","slug":"braven-improving-self-supervised-pre-training","title":"BRAVEn: Improving Self-Supervised Pre-training for Visual and Auditory Speech Recognition","date":"2024-04-02","arxiv_id":"2404.02098","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/braven-improving-self-supervised-pre-training#ran","syntology_url":"https://syntology.ai/paper/2404.02098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02098"}},"official":{"repos":["ahaliassos/raven"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/flowerformer-empowering-neural-architecture","slug":"flowerformer-empowering-neural-architecture","title":"FlowerFormer: Empowering Neural Architecture Encoding using a Flow-aware Graph Transformer","date":"2024-03-19","arxiv_id":"2403.12821","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/flowerformer-empowering-neural-architecture#ran","syntology_url":"https://syntology.ai/paper/2403.12821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12821"}},"official":{"repos":["y0ngjaenius/cvpr2024_flowerformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/speech-robust-bench-a-robustness-benchmark","slug":"speech-robust-bench-a-robustness-benchmark","title":"Speech Robust Bench: A Robustness Benchmark For Speech Recognition","date":"2024-03-08","arxiv_id":"2403.07937","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/speech-robust-bench-a-robustness-benchmark#ran","syntology_url":"https://syntology.ai/paper/2403.07937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07937"}},"official":{"repos":["ahmedshah1494/speech_robust_bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/where-visual-speech-meets-language-vsp-llm","slug":"where-visual-speech-meets-language-vsp-llm","title":"Where Visual Speech Meets Language: VSP-LLM Framework for Efficient and Context-Aware Visual Speech Processing","date":"2024-02-23","arxiv_id":"2402.15151","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":2,"n_ran_checked":7,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":12,"phrase":"10 ran (of which 2 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/where-visual-speech-meets-language-vsp-llm#ran","syntology_url":"https://syntology.ai/paper/2402.15151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15151"}},"official":{"repos":["sally-sh/vsp-llm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":2,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/how-do-hyenas-deal-with-human-speech-speech","slug":"how-do-hyenas-deal-with-human-speech-speech","title":"How do Hyenas deal with Human Speech? Speech Recognition and Translation with ConfHyena","date":"2024-02-20","arxiv_id":"2402.13208","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-do-hyenas-deal-with-human-speech-speech#ran","syntology_url":"https://syntology.ai/paper/2402.13208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13208"}},"official":{"repos":["hlt-mt/fbk-fairseq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/it-s-never-too-late-fusing-acoustic","slug":"it-s-never-too-late-fusing-acoustic","title":"It's Never Too Late: Fusing Acoustic Information into Large Language Models for Automatic Speech Recognition","date":"2024-02-08","arxiv_id":"2402.05457","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/it-s-never-too-late-fusing-acoustic#ran","syntology_url":"https://syntology.ai/paper/2402.05457","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05457"}},"official":null}},{"url":"/paper/unified-speech-text-pretraining-for-spoken","slug":"unified-speech-text-pretraining-for-spoken","title":"Paralinguistics-Aware Speech-Empowered Large Language Models for Natural Conversation","date":"2024-02-08","arxiv_id":"2402.05706","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":5,"n_ran_checked":7,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unified-speech-text-pretraining-for-spoken#ran","syntology_url":"https://syntology.ai/paper/2402.05706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05706"}},"official":{"repos":["naver-ai/usdm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/reborn-reinforcement-learned-boundary","slug":"reborn-reinforcement-learned-boundary","title":"REBORN: Reinforcement-Learned Boundary Segmentation with Iterative Training for Unsupervised ASR","date":"2024-02-06","arxiv_id":"2402.03988","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/reborn-reinforcement-learned-boundary#ran","syntology_url":"https://syntology.ai/paper/2402.03988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03988"}},"official":{"repos":["andybi7676/reborn-uasr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-efficient-learners","slug":"large-language-models-are-efficient-learners","title":"Large Language Models are Efficient Learners of Noise-Robust Speech Recognition","date":"2024-01-19","arxiv_id":"2401.10446","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-are-efficient-learners#ran","syntology_url":"https://syntology.ai/paper/2401.10446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10446"}},"official":{"repos":["yuchen005/robustger"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-online-sign-language-recognition-and","slug":"towards-online-sign-language-recognition-and","title":"Towards Online Continuous Sign Language Recognition and Translation","date":"2024-01-10","arxiv_id":"2401.05336","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-online-sign-language-recognition-and#ran","syntology_url":"https://syntology.ai/paper/2401.05336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05336"}},"official":{"repos":["FangyunWei/SLRT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multichannel-av-wav2vec2-a-framework-for","slug":"multichannel-av-wav2vec2-a-framework-for","title":"Multichannel AV-wav2vec2: A Framework for Learning Multichannel Multi-Modal Speech Representation","date":"2024-01-07","arxiv_id":"2401.03468","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":2,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"8 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multichannel-av-wav2vec2-a-framework-for#ran","syntology_url":"https://syntology.ai/paper/2401.03468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03468"}},"official":{"repos":["zqs01/multi-channel-wav2vec2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diarizationlm-speaker-diarization-post","slug":"diarizationlm-speaker-diarization-post","title":"DiarizationLM: Speaker Diarization Post-Processing with Large Language Models","date":"2024-01-07","arxiv_id":"2401.03506","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diarizationlm-speaker-diarization-post#ran","syntology_url":"https://syntology.ai/paper/2401.03506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03506"}},"official":{"repos":["google/speaker-id"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-for-autonomous-driving","slug":"large-language-models-for-autonomous-driving","title":"Personalized Autonomous Driving with Large Language Models: Field Experiments","date":"2023-12-14","arxiv_id":"2312.09397","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/large-language-models-for-autonomous-driving#ran","syntology_url":"https://syntology.ai/paper/2312.09397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09397"}},"official":null}},{"url":"/paper/graph-convolutions-enrich-the-self-attention","slug":"graph-convolutions-enrich-the-self-attention","title":"Graph Convolutions Enrich the Self-Attention in Transformers!","date":"2023-12-07","arxiv_id":"2312.04234","repositories_listed":1,"syntology":{"n":29,"n_ran":22,"n_constructed":0,"n_ran_checked":21,"n_instrument":1,"n_unverified":7,"n_honours":2,"n_violates":2,"n_no_contract":17,"n_pointer_only":10,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 21 with no instrument failure: 2 honoured, 2 violated, 17 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/graph-convolutions-enrich-the-self-attention#ran","syntology_url":"https://syntology.ai/paper/2312.04234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04234"}},"official":{"repos":["jeongwhanchoi/gfsa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/zero-shot-audio-captioning-with-audio","slug":"zero-shot-audio-captioning-with-audio","title":"Zero-shot audio captioning with audio-language model guidance and audio context keywords","date":"2023-11-14","arxiv_id":"2311.08396","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":17,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/zero-shot-audio-captioning-with-audio#ran","syntology_url":"https://syntology.ai/paper/2311.08396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08396"}},"official":{"repos":["explainableml/zeraucap"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/distil-whisper-robust-knowledge-distillation","slug":"distil-whisper-robust-knowledge-distillation","title":"Distil-Whisper: Robust Knowledge Distillation via Large-Scale Pseudo Labelling","date":"2023-11-01","arxiv_id":"2311.00430","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/distil-whisper-robust-knowledge-distillation#ran","syntology_url":"https://syntology.ai/paper/2311.00430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.00430"}},"official":{"repos":["huggingface/distil-whisper"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/torchaudio-2-1-advancing-speech-recognition","slug":"torchaudio-2-1-advancing-speech-recognition","title":"TorchAudio 2.1: Advancing speech recognition, self-supervised learning, and audio processing components for PyTorch","date":"2023-10-27","arxiv_id":"2310.17864","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/torchaudio-2-1-advancing-speech-recognition#ran","syntology_url":"https://syntology.ai/paper/2310.17864","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17864"}},"official":{"repos":["pytorch/audio"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/optimized-tokenization-for-transcribed-error","slug":"optimized-tokenization-for-transcribed-error","title":"Optimized Tokenization for Transcribed Error Correction","date":"2023-10-16","arxiv_id":"2310.10704","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/optimized-tokenization-for-transcribed-error#ran","syntology_url":"https://syntology.ai/paper/2310.10704","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10704"}},"official":null}},{"url":"/paper/advancing-test-time-adaptation-for-acoustic","slug":"advancing-test-time-adaptation-for-acoustic","title":"Advancing Test-Time Adaptation in Wild Acoustic Test Settings","date":"2023-10-14","arxiv_id":"2310.09505","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/advancing-test-time-adaptation-for-acoustic#ran","syntology_url":"https://syntology.ai/paper/2310.09505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09505"}},"official":{"repos":["Waffle-Liu/CEA"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/whispering-llama-a-cross-modal-generative","slug":"whispering-llama-a-cross-modal-generative","title":"Whispering LLaMA: A Cross-Modal Generative Error Correction Framework for Speech Recognition","date":"2023-10-10","arxiv_id":"2310.06434","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/whispering-llama-a-cross-modal-generative#ran","syntology_url":"https://syntology.ai/paper/2310.06434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06434"}},"official":{"repos":["srijith-rkr/whispering-llama"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lauragpt-listen-attend-understand-and","slug":"lauragpt-listen-attend-understand-and","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","date":"2023-10-07","arxiv_id":"2310.04673","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lauragpt-listen-attend-understand-and#ran","syntology_url":"https://syntology.ai/paper/2310.04673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04673"}},"official":null}},{"url":"/paper/librispeech-pc-benchmark-for-evaluation-of","slug":"librispeech-pc-benchmark-for-evaluation-of","title":"LibriSpeech-PC: Benchmark for Evaluation of Punctuation and Capitalization Capabilities of end-to-end ASR Models","date":"2023-10-04","arxiv_id":"2310.02943","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/librispeech-pc-benchmark-for-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2310.02943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02943"}},"official":null}},{"url":"/paper/unsupervised-speech-recognition-with-n","slug":"unsupervised-speech-recognition-with-n","title":"Unsupervised Speech Recognition with N-Skipgram and Positional Unigram Matching","date":"2023-10-03","arxiv_id":"2310.02382","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-speech-recognition-with-n#ran","syntology_url":"https://syntology.ai/paper/2310.02382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02382"}},"official":{"repos":["lwang114/graphunsupasr"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/federated-learning-with-differential-privacy","slug":"federated-learning-with-differential-privacy","title":"Enabling Differentially Private Federated Learning for Speech Recognition: Benchmarks, Adaptive Optimizers and Gradient Clipping","date":"2023-09-29","arxiv_id":"2310.00098","repositories_listed":0,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/federated-learning-with-differential-privacy#ran","syntology_url":"https://syntology.ai/paper/2310.00098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00098"}},"official":null}},{"url":"/paper/hyporadise-an-open-baseline-for-generative-1","slug":"hyporadise-an-open-baseline-for-generative-1","title":"HyPoradise: An Open Baseline for Generative Speech Recognition with Large Language Models","date":"2023-09-27","arxiv_id":"2309.15701","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hyporadise-an-open-baseline-for-generative-1#ran","syntology_url":"https://syntology.ai/paper/2309.15701","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15701"}},"official":{"repos":["hypotheses-paradise/hypo2trans"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/comflp-correlation-measure-based-fast-search","slug":"comflp-correlation-measure-based-fast-search","title":"CoMFLP: Correlation Measure based Fast Search on ASR Layer Pruning","date":"2023-09-21","arxiv_id":"2309.11768","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/comflp-correlation-measure-based-fast-search#ran","syntology_url":"https://syntology.ai/paper/2309.11768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11768"}},"official":{"repos":["louislau1129/comflp"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/encodecmae-leveraging-neural-codecs-for","slug":"encodecmae-leveraging-neural-codecs-for","title":"EnCodecMAE: Leveraging neural codecs for universal audio representation learning","date":"2023-09-14","arxiv_id":"2309.07391","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/encodecmae-leveraging-neural-codecs-for#ran","syntology_url":"https://syntology.ai/paper/2309.07391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07391"}},"official":{"repos":["habla-liaa/encodecmae"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/active-learning-for-classifying-2d-grid-based","slug":"active-learning-for-classifying-2d-grid-based","title":"Active Learning for Classifying 2D Grid-Based Level Completability","date":"2023-09-08","arxiv_id":"2309.04367","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/active-learning-for-classifying-2d-grid-based#ran","syntology_url":"https://syntology.ai/paper/2309.04367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.04367"}},"official":{"repos":["mahsabazzaz/level-completabilty-x-active-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/blsp-bootstrapping-language-speech-pre-1","slug":"blsp-bootstrapping-language-speech-pre-1","title":"BLSP: Bootstrapping Language-Speech Pre-training via Behavior Alignment of Continuation Writing","date":"2023-09-02","arxiv_id":"2309.00916","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/blsp-bootstrapping-language-speech-pre-1#ran","syntology_url":"https://syntology.ai/paper/2309.00916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.00916"}},"official":{"repos":["cwang621/blsp"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lip2vec-efficient-and-robust-visual-speech","slug":"lip2vec-efficient-and-robust-visual-speech","title":"Lip2Vec: Efficient and Robust Visual Speech Recognition via Latent-to-Latent Visual to Audio Representation Mapping","date":"2023-08-11","arxiv_id":"2308.06112","repositories_listed":0,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":16,"n_pointer_only":19,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 1 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lip2vec-efficient-and-robust-visual-speech#ran","syntology_url":"https://syntology.ai/paper/2308.06112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06112"}},"official":null}},{"url":"/paper/towards-stealthy-backdoor-attacks-against","slug":"towards-stealthy-backdoor-attacks-against","title":"Towards Stealthy Backdoor Attacks against Speech Recognition via Elements of Sound","date":"2023-07-17","arxiv_id":"2307.08208","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-stealthy-backdoor-attacks-against#ran","syntology_url":"https://syntology.ai/paper/2307.08208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08208"}},"official":{"repos":["hanbocai/badspeech_soe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/quilt-1m-one-million-image-text-pairs-for-1","slug":"quilt-1m-one-million-image-text-pairs-for-1","title":"Quilt-1M: One Million Image-Text Pairs for Histopathology","date":"2023-06-20","arxiv_id":"2306.11207","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quilt-1m-one-million-image-text-pairs-for-1#ran","syntology_url":"https://syntology.ai/paper/2306.11207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11207"}},"official":{"repos":["wisdomikezogwo/quilt1m"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hearing-lips-in-noise-universal-viseme","slug":"hearing-lips-in-noise-universal-viseme","title":"Hearing Lips in Noise: Universal Viseme-Phoneme Mapping and Transfer for Robust Audio-Visual Speech Recognition","date":"2023-06-18","arxiv_id":"2306.10563","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hearing-lips-in-noise-universal-viseme#ran","syntology_url":"https://syntology.ai/paper/2306.10563","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10563"}},"official":{"repos":["yuchen005/univpm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mir-gan-refining-frame-level-modality","slug":"mir-gan-refining-frame-level-modality","title":"MIR-GAN: Refining Frame-Level Modality-Invariant Representations with Adversarial Network for Audio-Visual Speech Recognition","date":"2023-06-18","arxiv_id":"2306.10567","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/mir-gan-refining-frame-level-modality#ran","syntology_url":"https://syntology.ai/paper/2306.10567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10567"}},"official":{"repos":["yuchen005/mir-gan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sgem-test-time-adaptation-for-automatic","slug":"sgem-test-time-adaptation-for-automatic","title":"SGEM: Test-Time Adaptation for Automatic Speech Recognition via Sequential-Level Generalized Entropy Minimization","date":"2023-06-03","arxiv_id":"2306.01981","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sgem-test-time-adaptation-for-automatic#ran","syntology_url":"https://syntology.ai/paper/2306.01981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01981"}},"official":{"repos":["drumpt/sgem"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/improved-deepfake-detection-using-whisper","slug":"improved-deepfake-detection-using-whisper","title":"Improved DeepFake Detection Using Whisper Features","date":"2023-06-02","arxiv_id":"2306.01428","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improved-deepfake-detection-using-whisper#ran","syntology_url":"https://syntology.ai/paper/2306.01428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01428"}},"official":{"repos":["piotrkawa/deepfake-whisper-features"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/perception-and-semantic-aware-regularization-1","slug":"perception-and-semantic-aware-regularization-1","title":"Perception and Semantic Aware Regularization for Sequential Confidence Calibration","date":"2023-05-31","arxiv_id":"2305.19498","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/perception-and-semantic-aware-regularization-1#ran","syntology_url":"https://syntology.ai/paper/2305.19498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19498"}},"official":{"repos":["husterpzh/pssr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-speech-technology-to-1000-languages-1","slug":"scaling-speech-technology-to-1000-languages-1","title":"Scaling Speech Technology to 1,000+ Languages","date":"2023-05-22","arxiv_id":"2305.13516","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-speech-technology-to-1000-languages-1#ran","syntology_url":"https://syntology.ai/paper/2305.13516","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13516"}},"official":{"repos":["facebookresearch/fairseq","pytorch/fairseq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/making-more-of-little-data-improving-low","slug":"making-more-of-little-data-improving-low","title":"Making More of Little Data: Improving Low-Resource Automatic Speech Recognition Using Data Augmentation","date":"2023-05-18","arxiv_id":"2305.10951","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-more-of-little-data-improving-low#ran","syntology_url":"https://syntology.ai/paper/2305.10951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10951"}},"official":{"repos":["bartelds/asr-augmentation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-fine-tuning-for-improved","slug":"self-supervised-fine-tuning-for-improved","title":"Self-supervised Fine-tuning for Improved Content Representations by Speaker-invariant Clustering","date":"2023-05-18","arxiv_id":"2305.11072","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-supervised-fine-tuning-for-improved#ran","syntology_url":"https://syntology.ai/paper/2305.11072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11072"}},"official":{"repos":["vectominist/spin"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/x-llm-bootstrapping-advanced-large-language","slug":"x-llm-bootstrapping-advanced-large-language","title":"X-LLM: Bootstrapping Advanced Large Language Models by Treating Multi-Modalities as Foreign Languages","date":"2023-05-07","arxiv_id":"2305.04160","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/x-llm-bootstrapping-advanced-large-language#ran","syntology_url":"https://syntology.ai/paper/2305.04160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04160"}},"official":null}},{"url":"/paper/auto-avsr-audio-visual-speech-recognition","slug":"auto-avsr-audio-visual-speech-recognition","title":"Auto-AVSR: Audio-Visual Speech Recognition with Automatic Labels","date":"2023-03-25","arxiv_id":"2303.14307","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/auto-avsr-audio-visual-speech-recognition#ran","syntology_url":"https://syntology.ai/paper/2303.14307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14307"}},"official":{"repos":["mpc001/auto_avsr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/watch-or-listen-robust-audio-visual-speech","slug":"watch-or-listen-robust-audio-visual-speech","title":"Watch or Listen: Robust Audio-Visual Speech Recognition with Visual Corruption Modeling and Reliability Scoring","date":"2023-03-15","arxiv_id":"2303.08536","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/watch-or-listen-robust-audio-visual-speech#ran","syntology_url":"https://syntology.ai/paper/2303.08536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08536"}},"official":{"repos":["ms-dot-k/AVSR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stabilizing-transformer-training-by","slug":"stabilizing-transformer-training-by","title":"Stabilizing Transformer Training by Preventing Attention Entropy Collapse","date":"2023-03-11","arxiv_id":"2303.06296","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stabilizing-transformer-training-by#ran","syntology_url":"https://syntology.ai/paper/2303.06296","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06296"}},"official":{"repos":["apple/ml-sigma-reparam"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/calibrating-transformers-via-sparse-gaussian","slug":"calibrating-transformers-via-sparse-gaussian","title":"Calibrating Transformers via Sparse Gaussian Processes","date":"2023-03-04","arxiv_id":"2303.02444","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/calibrating-transformers-via-sparse-gaussian#ran","syntology_url":"https://syntology.ai/paper/2303.02444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.02444"}},"official":{"repos":["chenw20/sgpa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/brainbert-self-supervised-representation","slug":"brainbert-self-supervised-representation","title":"BrainBERT: Self-supervised representation learning for intracranial recordings","date":"2023-02-28","arxiv_id":"2302.14367","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":4,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/brainbert-self-supervised-representation#ran","syntology_url":"https://syntology.ai/paper/2302.14367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.14367"}},"official":{"repos":["czlwang/brainbert"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/complex-dynamic-neurons-improved-spiking","slug":"complex-dynamic-neurons-improved-spiking","title":"Complex Dynamic Neurons Improved Spiking Transformer Network for Efficient Automatic Speech Recognition","date":"2023-02-02","arxiv_id":"2302.01194","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/complex-dynamic-neurons-improved-spiking#ran","syntology_url":"https://syntology.ai/paper/2302.01194","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.01194"}},"official":{"repos":["MingLunHan/CIF-PyTorch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/olkavs-an-open-large-scale-korean-audio","slug":"olkavs-an-open-large-scale-korean-audio","title":"OLKAVS: An Open Large-Scale Korean Audio-Visual Speech Dataset","date":"2023-01-16","arxiv_id":"2301.06375","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/olkavs-an-open-large-scale-korean-audio#ran","syntology_url":"https://syntology.ai/paper/2301.06375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.06375"}},"official":{"repos":["iip-sogang/olkavs-avspeech"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/audio-visual-efficient-conformer-for-robust","slug":"audio-visual-efficient-conformer-for-robust","title":"Audio-Visual Efficient Conformer for Robust Speech Recognition","date":"2023-01-04","arxiv_id":"2301.01456","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/audio-visual-efficient-conformer-for-robust#ran","syntology_url":"https://syntology.ai/paper/2301.01456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.01456"}},"official":{"repos":["burchim/avec"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-voice-reconstruction-from-eeg-during","slug":"towards-voice-reconstruction-from-eeg-during","title":"Towards Voice Reconstruction from EEG during Imagined Speech","date":"2023-01-02","arxiv_id":"2301.07173","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-voice-reconstruction-from-eeg-during#ran","syntology_url":"https://syntology.ai/paper/2301.07173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07173"}},"official":{"repos":["youngeun1209/neurotalk"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-self-supervised-learning-with","slug":"efficient-self-supervised-learning-with","title":"Efficient Self-supervised Learning with Contextualized Target Representations for Vision, Speech and Language","date":"2022-12-14","arxiv_id":"2212.07525","repositories_listed":5,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/efficient-self-supervised-learning-with#ran","syntology_url":"https://syntology.ai/paper/2212.07525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07525"}},"official":{"repos":["facebookresearch/fairseq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/robust-speech-recognition-via-large-scale-1","slug":"robust-speech-recognition-via-large-scale-1","title":"Robust Speech Recognition via Large-Scale Weak Supervision","date":"2022-12-06","arxiv_id":"2212.04356","repositories_listed":15,"syntology":{"n":59,"n_ran":49,"n_constructed":0,"n_ran_checked":47,"n_instrument":2,"n_unverified":10,"n_honours":2,"n_violates":0,"n_no_contract":45,"n_pointer_only":21,"phrase":"49 ran (of which 0 constructed an object rather than computing a result; 47 with no instrument failure: 2 honoured, 0 violated, 45 with no contract checked; 2 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/robust-speech-recognition-via-large-scale-1#ran","syntology_url":"https://syntology.ai/paper/2212.04356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.04356"}},"official":{"repos":["openai/whisper"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/softcorrect-error-correction-with-soft","slug":"softcorrect-error-correction-with-soft","title":"SoftCorrect: Error Correction with Soft Detection for Automatic Speech Recognition","date":"2022-12-02","arxiv_id":"2212.01039","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/softcorrect-error-correction-with-soft#ran","syntology_url":"https://syntology.ai/paper/2212.01039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.01039"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/euro-espnet-unsupervised-asr-open-source","slug":"euro-espnet-unsupervised-asr-open-source","title":"EURO: ESPnet Unsupervised ASR Open-source Toolkit","date":"2022-11-30","arxiv_id":"2211.17196","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/euro-espnet-unsupervised-asr-open-source#ran","syntology_url":"https://syntology.ai/paper/2211.17196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.17196"}},"official":{"repos":["espnet/espnet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-we-use-common-voice-to-train-a-multi","slug":"can-we-use-common-voice-to-train-a-multi","title":"Can we use Common Voice to train a Multi-Speaker TTS system?","date":"2022-10-12","arxiv_id":"2210.06370","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-we-use-common-voice-to-train-a-multi#ran","syntology_url":"https://syntology.ai/paper/2210.06370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06370"}},"official":null}},{"url":"/paper/joeys2t-minimalistic-speech-to-text-modeling","slug":"joeys2t-minimalistic-speech-to-text-modeling","title":"JoeyS2T: Minimalistic Speech-to-Text Modeling with JoeyNMT","date":"2022-10-05","arxiv_id":"2210.02545","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/joeys2t-minimalistic-speech-to-text-modeling#ran","syntology_url":"https://syntology.ai/paper/2210.02545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.02545"}},"official":{"repos":["may-/joeys2t"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/indicsuperb-a-speech-processing-universal","slug":"indicsuperb-a-speech-processing-universal","title":"IndicSUPERB: A Speech Processing Universal Performance Benchmark for Indian languages","date":"2022-08-24","arxiv_id":"2208.11761","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/indicsuperb-a-speech-processing-universal#ran","syntology_url":"https://syntology.ai/paper/2208.11761","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.11761"}},"official":{"repos":["AI4Bharat/indicSUPERB"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-transfer-learning-of-wav2vec-2-0-for","slug":"towards-transfer-learning-of-wav2vec-2-0-for","title":"Transfer Learning of wav2vec 2.0 for Automatic Lyric Transcription","date":"2022-07-20","arxiv_id":"2207.09747","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-transfer-learning-of-wav2vec-2-0-for#ran","syntology_url":"https://syntology.ai/paper/2207.09747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.09747"}},"official":{"repos":["guxm2021/alt_speechbrain"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-understanding-and-mitigating-audio","slug":"towards-understanding-and-mitigating-audio","title":"Towards Understanding and Mitigating Audio Adversarial Examples for Speaker Recognition","date":"2022-06-07","arxiv_id":"2206.03393","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-understanding-and-mitigating-audio#ran","syntology_url":"https://syntology.ai/paper/2206.03393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.03393"}},"official":null}},{"url":"/paper/variable-rate-hierarchical-cpc-leads-to","slug":"variable-rate-hierarchical-cpc-leads-to","title":"Variable-rate hierarchical CPC leads to acoustic unit discovery in speech","date":"2022-06-05","arxiv_id":"2206.02211","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/variable-rate-hierarchical-cpc-leads-to#ran","syntology_url":"https://syntology.ai/paper/2206.02211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.02211"}},"official":{"repos":["chorowski-lab/hcpc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-scale-streaming-end-to-end-speech","slug":"large-scale-streaming-end-to-end-speech","title":"Large-Scale Streaming End-to-End Speech Translation with Neural Transducers","date":"2022-04-11","arxiv_id":"2204.05352","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-scale-streaming-end-to-end-speech#ran","syntology_url":"https://syntology.ai/paper/2204.05352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.05352"}},"official":null}},{"url":"/paper/unsupervised-uncertainty-measures-of","slug":"unsupervised-uncertainty-measures-of","title":"Unsupervised Uncertainty Measures of Automatic Speech Recognition for Non-intrusive Speech Intelligibility Prediction","date":"2022-04-08","arxiv_id":"2204.04288","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsupervised-uncertainty-measures-of#ran","syntology_url":"https://syntology.ai/paper/2204.04288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.04288"}},"official":{"repos":["claritychallenge/clarity"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/3m-multi-loss-multi-path-and-multi-level","slug":"3m-multi-loss-multi-path-and-multi-level","title":"3M: Multi-loss, Multi-path and Multi-level Neural Networks for speech recognition","date":"2022-04-07","arxiv_id":"2204.03178","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3m-multi-loss-multi-path-and-multi-level#ran","syntology_url":"https://syntology.ai/paper/2204.03178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.03178"}},"official":{"repos":["tencent-ailab/3m-asr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wenet-2-0-more-productive-end-to-end-speech","slug":"wenet-2-0-more-productive-end-to-end-speech","title":"WeNet 2.0: More Productive End-to-End Speech Recognition Toolkit","date":"2022-03-29","arxiv_id":"2203.15455","repositories_listed":3,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":9,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wenet-2-0-more-productive-end-to-end-speech#ran","syntology_url":"https://syntology.ai/paper/2203.15455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15455"}},"official":{"repos":["wenet-e2e/wenet"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/lighthubert-lightweight-and-configurable","slug":"lighthubert-lightweight-and-configurable","title":"LightHuBERT: Lightweight and Configurable Speech Representation Learning with Once-for-All Hidden-Unit BERT","date":"2022-03-29","arxiv_id":"2203.15610","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/lighthubert-lightweight-and-configurable#ran","syntology_url":"https://syntology.ai/paper/2203.15610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15610"}},"official":{"repos":["mechanicalsea/lighthubert"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/recent-improvements-of-asr-models-in-the-face","slug":"recent-improvements-of-asr-models-in-the-face","title":"Recent improvements of ASR models in the face of adversarial attacks","date":"2022-03-29","arxiv_id":"2203.16536","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recent-improvements-of-asr-models-in-the-face#ran","syntology_url":"https://syntology.ai/paper/2203.16536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16536"}},"official":{"repos":["raphaelolivier/robust_speech"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cmgan-conformer-based-metric-gan-for-speech","slug":"cmgan-conformer-based-metric-gan-for-speech","title":"CMGAN: Conformer-based Metric GAN for Speech Enhancement","date":"2022-03-28","arxiv_id":"2203.15149","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":2,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 3 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cmgan-conformer-based-metric-gan-for-speech#ran","syntology_url":"https://syntology.ai/paper/2203.15149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15149"}},"official":{"repos":["ruizhecao96/cmgan"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flute-a-scalable-extensible-framework-for","slug":"flute-a-scalable-extensible-framework-for","title":"FLUTE: A Scalable, Extensible Framework for High-Performance Federated Learning Simulations","date":"2022-03-25","arxiv_id":"2203.13789","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flute-a-scalable-extensible-framework-for#ran","syntology_url":"https://syntology.ai/paper/2203.13789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13789"}},"official":{"repos":["microsoft/msrflute"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-3-t-alignment-aware-acoustic-and-text","slug":"a-3-t-alignment-aware-acoustic-and-text","title":"A$^3$T: Alignment-Aware Acoustic and Text Pretraining for Speech Synthesis and Editing","date":"2022-03-18","arxiv_id":"2203.09690","repositories_listed":2,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/a-3-t-alignment-aware-acoustic-and-text#ran","syntology_url":"https://syntology.ai/paper/2203.09690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09690"}},"official":null}}],"record_sha256":"b998af2093668bd122fb529e0ba507a27322a9a3f777493fe089dacb7a5c9bd4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}