{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/7","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":58,"rows_per_page":100,"rows":[601,700],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/6","next":"/task/speech-recognition-1/papers/8","papers":[{"url":"/paper/arabic-dysarthric-speech-recognition-using","slug":"arabic-dysarthric-speech-recognition-using","title":"Arabic Dysarthric Speech Recognition Using Adversarial and Signal-Based Augmentation","date":"2023-06-07","arxiv_id":"2306.04368","repositories_listed":1,"syntology":null},{"url":"/paper/zambezi-voice-a-multilingual-speech-corpus","slug":"zambezi-voice-a-multilingual-speech-corpus","title":"Zambezi Voice: A Multilingual Speech Corpus for Zambian Languages","date":"2023-06-07","arxiv_id":"2306.04428","repositories_listed":1,"syntology":null},{"url":"/paper/mavd-the-first-open-large-scale-mandarin","slug":"mavd-the-first-open-large-scale-mandarin","title":"MAVD: The First Open Large-Scale Mandarin Audio-Visual Dataset with Depth Information","date":"2023-06-04","arxiv_id":"2306.02263","repositories_listed":1,"syntology":null},{"url":"/paper/spellmapper-a-non-autoregressive-neural","slug":"spellmapper-a-non-autoregressive-neural","title":"SpellMapper: A non-autoregressive neural spellchecker for ASR customization with candidate retrieval based on n-gram mappings","date":"2023-06-04","arxiv_id":"2306.02317","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-pretrained-asr-models-to-low","slug":"adapting-pretrained-asr-models-to-low","title":"Advancing African-Accented Speech Recognition: Epistemic Uncertainty-Driven Data Selection for Generalizable ASR Models","date":"2023-06-03","arxiv_id":"2306.02105","repositories_listed":1,"syntology":null},{"url":"/paper/sgem-test-time-adaptation-for-automatic","slug":"sgem-test-time-adaptation-for-automatic","title":"SGEM: Test-Time Adaptation for Automatic Speech Recognition via Sequential-Level Generalized Entropy Minimization","date":"2023-06-03","arxiv_id":"2306.01981","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sgem-test-time-adaptation-for-automatic#ran","syntology_url":"https://syntology.ai/paper/2306.01981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01981"}},"official":{"repos":["drumpt/sgem"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-contextual-biasing-remain-effective-with","slug":"can-contextual-biasing-remain-effective-with","title":"Can Contextual Biasing Remain Effective with Whisper and GPT-2?","date":"2023-06-02","arxiv_id":"2306.01942","repositories_listed":1,"syntology":null},{"url":"/paper/distilxlsr-a-light-weight-cross-lingual","slug":"distilxlsr-a-light-weight-cross-lingual","title":"DistilXLSR: A Light Weight Cross-Lingual Speech Representation Model","date":"2023-06-02","arxiv_id":"2306.01303","repositories_listed":1,"syntology":null},{"url":"/paper/explainability-of-speech-recognition","slug":"explainability-of-speech-recognition","title":"Explainability of Speech Recognition Transformers via Gradient-based Attention Visualization","date":"2023-06-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/improved-deepfake-detection-using-whisper","slug":"improved-deepfake-detection-using-whisper","title":"Improved DeepFake Detection Using Whisper Features","date":"2023-06-02","arxiv_id":"2306.01428","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improved-deepfake-detection-using-whisper#ran","syntology_url":"https://syntology.ai/paper/2306.01428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01428"}},"official":{"repos":["piotrkawa/deepfake-whisper-features"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/slothspeech-denial-of-service-attack-against","slug":"slothspeech-denial-of-service-attack-against","title":"SlothSpeech: Denial-of-service Attack Against Speech Recognition Models","date":"2023-06-01","arxiv_id":"2306.00794","repositories_listed":1,"syntology":null},{"url":"/paper/perception-and-semantic-aware-regularization-1","slug":"perception-and-semantic-aware-regularization-1","title":"Perception and Semantic Aware Regularization for Sequential Confidence Calibration","date":"2023-05-31","arxiv_id":"2305.19498","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/perception-and-semantic-aware-regularization-1#ran","syntology_url":"https://syntology.ai/paper/2305.19498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19498"}},"official":{"repos":["husterpzh/pssr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/graph-neural-networks-for-contextual-asr-with","slug":"graph-neural-networks-for-contextual-asr-with","title":"Graph Neural Networks for Contextual ASR with the Tree-Constrained Pointer Generator","date":"2023-05-30","arxiv_id":"2305.18824","repositories_listed":1,"syntology":null},{"url":"/paper/commonaccent-exploring-large-acoustic","slug":"commonaccent-exploring-large-acoustic","title":"CommonAccent: Exploring Large Acoustic Pretrained Models for Accent Classification Based on Common Voice","date":"2023-05-29","arxiv_id":"2305.18283","repositories_listed":1,"syntology":null},{"url":"/paper/hyperconformer-multi-head-hypermixer-for","slug":"hyperconformer-multi-head-hypermixer-for","title":"HyperConformer: Multi-head HyperMixer for Efficient Speech Recognition","date":"2023-05-29","arxiv_id":"2305.18281","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-granularity-gap-for-acoustic","slug":"bridging-the-granularity-gap-for-acoustic","title":"Bridging the Granularity Gap for Acoustic Modeling","date":"2023-05-27","arxiv_id":"2305.17356","repositories_listed":1,"syntology":null},{"url":"/paper/big-c-a-multimodal-multi-purpose-dataset-for","slug":"big-c-a-multimodal-multi-purpose-dataset-for","title":"BIG-C: a Multimodal Multi-Purpose Dataset for Bemba","date":"2023-05-26","arxiv_id":"2305.17202","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-characteristics-of-the-output","slug":"leveraging-characteristics-of-the-output","title":"DistriBlock: Identifying adversarial audio samples by leveraging characteristics of the output distribution","date":"2023-05-26","arxiv_id":"2305.17000","repositories_listed":1,"syntology":null},{"url":"/paper/copyne-better-contextual-asr-by-copying-named","slug":"copyne-better-contextual-asr-by-copying-named","title":"CopyNE: Better Contextual ASR by Copying Named Entities","date":"2023-05-22","arxiv_id":"2305.12839","repositories_listed":1,"syntology":null},{"url":"/paper/blank-regularized-ctc-for-frame-skipping-in","slug":"blank-regularized-ctc-for-frame-skipping-in","title":"Blank-regularized CTC for Frame Skipping in Neural Transducer","date":"2023-05-19","arxiv_id":"2305.11558","repositories_listed":1,"syntology":null},{"url":"/paper/funasr-a-fundamental-end-to-end-speech","slug":"funasr-a-fundamental-end-to-end-speech","title":"FunASR: A Fundamental End-to-End Speech Recognition Toolkit","date":"2023-05-18","arxiv_id":"2305.11013","repositories_listed":1,"syntology":null},{"url":"/paper/making-more-of-little-data-improving-low","slug":"making-more-of-little-data-improving-low","title":"Making More of Little Data: Improving Low-Resource Automatic Speech Recognition Using Data Augmentation","date":"2023-05-18","arxiv_id":"2305.10951","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/making-more-of-little-data-improving-low#ran","syntology_url":"https://syntology.ai/paper/2305.10951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10951"}},"official":{"repos":["bartelds/asr-augmentation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompting-the-hidden-talent-of-web-scale","slug":"prompting-the-hidden-talent-of-web-scale","title":"Prompting the Hidden Talent of Web-Scale Speech Models for Zero-Shot Task Generalization","date":"2023-05-18","arxiv_id":"2305.11095","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-fine-tuning-for-improved","slug":"self-supervised-fine-tuning-for-improved","title":"Self-supervised Fine-tuning for Improved Content Representations by Speaker-invariant Clustering","date":"2023-05-18","arxiv_id":"2305.11072","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-supervised-fine-tuning-for-improved#ran","syntology_url":"https://syntology.ai/paper/2305.11072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11072"}},"official":{"repos":["vectominist/spin"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-global-interaction-and-local","slug":"cross-modal-global-interaction-and-local","title":"Cross-Modal Global Interaction and Local Alignment for Audio-Visual Speech Recognition","date":"2023-05-16","arxiv_id":"2305.09212","repositories_listed":1,"syntology":null},{"url":"/paper/back-translation-for-speech-to-text","slug":"back-translation-for-speech-to-text","title":"Back Translation for Speech-to-text Translation Without Transcripts","date":"2023-05-15","arxiv_id":"2305.08709","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-variants-of-wav2vec-2-0-on","slug":"evaluating-variants-of-wav2vec-2-0-on","title":"Evaluating Variants of wav2vec 2.0 on Affective Vocal Burst Tasks","date":"2023-05-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towards-better-domain-adaptation-for-self","slug":"towards-better-domain-adaptation-for-self","title":"Towards Better Domain Adaptation for Self-supervised Models: A Case Study of Child ASR","date":"2023-04-28","arxiv_id":"2305.00115","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-deep-learning-models-for-raspberry","slug":"optimizing-deep-learning-models-for-raspberry","title":"Optimizing Deep Learning Models For Raspberry Pi","date":"2023-04-25","arxiv_id":"2304.13039","repositories_listed":1,"syntology":null},{"url":"/paper/olisia-a-cascade-system-for-spoken-dialogue","slug":"olisia-a-cascade-system-for-spoken-dialogue","title":"OLISIA: a Cascade System for Spoken Dialogue State Tracking","date":"2023-04-20","arxiv_id":"2304.11073","repositories_listed":1,"syntology":null},{"url":"/paper/cb-conformer-contextual-biasing-conformer-for","slug":"cb-conformer-contextual-biasing-conformer-for","title":"CB-Conformer: Contextual biasing Conformer for biased word recognition","date":"2023-04-19","arxiv_id":"2304.09607","repositories_listed":1,"syntology":null},{"url":"/paper/political-corpus-creation-through-automatic","slug":"political-corpus-creation-through-automatic","title":"Political corpus creation through automatic speech recognition on EU debates","date":"2023-04-17","arxiv_id":"2304.08137","repositories_listed":1,"syntology":null},{"url":"/paper/acoustic-absement-in-detail-quantifying","slug":"acoustic-absement-in-detail-quantifying","title":"Acoustic absement in detail: Quantifying acoustic differences across time-series representations of speech data","date":"2023-04-12","arxiv_id":"2304.06183","repositories_listed":1,"syntology":null},{"url":"/paper/certifiable-black-box-attack-ensuring","slug":"certifiable-black-box-attack-ensuring","title":"Certifiable Black-Box Attacks with Randomized Adversarial Examples: Breaking Defenses with Provable Confidence","date":"2023-04-10","arxiv_id":"2304.04343","repositories_listed":1,"syntology":null},{"url":"/paper/cico-domain-aware-sign-language-retrieval-via","slug":"cico-domain-aware-sign-language-retrieval-via","title":"CiCo: Domain-Aware Sign Language Retrieval via Cross-Lingual Contrastive Learning","date":"2023-03-22","arxiv_id":"2303.12793","repositories_listed":1,"syntology":null},{"url":"/paper/cascading-and-direct-approaches-to","slug":"cascading-and-direct-approaches-to","title":"Cascading and Direct Approaches to Unsupervised Constituency Parsing on Spoken Sentences","date":"2023-03-15","arxiv_id":"2303.08809","repositories_listed":1,"syntology":null},{"url":"/paper/hybridformer-improving-squeezeformer-with","slug":"hybridformer-improving-squeezeformer-with","title":"HYBRIDFORMER: improving SqueezeFormer with hybrid attention and NSR mechanism","date":"2023-03-15","arxiv_id":"2303.08636","repositories_listed":1,"syntology":null},{"url":"/paper/watch-or-listen-robust-audio-visual-speech","slug":"watch-or-listen-robust-audio-visual-speech","title":"Watch or Listen: Robust Audio-Visual Speech Recognition with Visual Corruption Modeling and Reliability Scoring","date":"2023-03-15","arxiv_id":"2303.08536","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/watch-or-listen-robust-audio-visual-speech#ran","syntology_url":"https://syntology.ai/paper/2303.08536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08536"}},"official":{"repos":["ms-dot-k/AVSR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/i3d-transformer-architectures-with-input","slug":"i3d-transformer-architectures-with-input","title":"I3D: Transformer architectures with input-dependent dynamic depth for speech recognition","date":"2023-03-14","arxiv_id":"2303.07624","repositories_listed":1,"syntology":null},{"url":"/paper/fine-tuning-strategies-for-faster-inference","slug":"fine-tuning-strategies-for-faster-inference","title":"Fine-tuning Strategies for Faster Inference using Speech Self-Supervised Models: A Comparative Study","date":"2023-03-12","arxiv_id":"2303.06740","repositories_listed":1,"syntology":null},{"url":"/paper/stabilizing-transformer-training-by","slug":"stabilizing-transformer-training-by","title":"Stabilizing Transformer Training by Preventing Attention Entropy Collapse","date":"2023-03-11","arxiv_id":"2303.06296","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stabilizing-transformer-training-by#ran","syntology_url":"https://syntology.ai/paper/2303.06296","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06296"}},"official":{"repos":["apple/ml-sigma-reparam"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transcription-free-filler-word-detection-with","slug":"transcription-free-filler-word-detection-with","title":"Transcription free filler word detection with Neural semi-CRFs","date":"2023-03-11","arxiv_id":"2303.06475","repositories_listed":1,"syntology":null},{"url":"/paper/deepgd-a-multi-objective-black-box-test","slug":"deepgd-a-multi-objective-black-box-test","title":"DeepGD: A Multi-Objective Black-Box Test Selection Approach for Deep Neural Networks","date":"2023-03-08","arxiv_id":"2303.04878","repositories_listed":1,"syntology":null},{"url":"/paper/calibrating-transformers-via-sparse-gaussian","slug":"calibrating-transformers-via-sparse-gaussian","title":"Calibrating Transformers via Sparse Gaussian Processes","date":"2023-03-04","arxiv_id":"2303.02444","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/calibrating-transformers-via-sparse-gaussian#ran","syntology_url":"https://syntology.ai/paper/2303.02444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.02444"}},"official":{"repos":["chenw20/sgpa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/muavic-a-multilingual-audio-visual-corpus-for","slug":"muavic-a-multilingual-audio-visual-corpus-for","title":"MuAViC: A Multilingual Audio-Visual Corpus for Robust Speech Recognition and Robust Speech-to-Text Translation","date":"2023-03-01","arxiv_id":"2303.00628","repositories_listed":1,"syntology":null},{"url":"/paper/brainbert-self-supervised-representation","slug":"brainbert-self-supervised-representation","title":"BrainBERT: Self-supervised representation learning for intracranial recordings","date":"2023-02-28","arxiv_id":"2302.14367","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":4,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/brainbert-self-supervised-representation#ran","syntology_url":"https://syntology.ai/paper/2302.14367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.14367"}},"official":{"repos":["czlwang/brainbert"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/language-universal-adapter-learning-with","slug":"language-universal-adapter-learning-with","title":"Language-Universal Adapter Learning with Knowledge Distillation for End-to-End Multilingual Speech Recognition","date":"2023-02-28","arxiv_id":"2303.01249","repositories_listed":1,"syntology":null},{"url":"/paper/low-latency-transformers-for-speech","slug":"low-latency-transformers-for-speech","title":"A low latency attention module for streaming self-supervised speech representation learning","date":"2023-02-27","arxiv_id":"2302.13451","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-speech-recognition-for-language","slug":"multimodal-speech-recognition-for-language","title":"Multimodal Speech Recognition for Language-Guided Embodied Agents","date":"2023-02-27","arxiv_id":"2302.14030","repositories_listed":1,"syntology":null},{"url":"/paper/structured-pruning-of-self-supervised-pre","slug":"structured-pruning-of-self-supervised-pre","title":"Structured Pruning of Self-Supervised Pre-trained Models for Speech Recognition and Understanding","date":"2023-02-27","arxiv_id":"2302.14132","repositories_listed":1,"syntology":null},{"url":"/paper/text-only-domain-adaptation-for-end-to-end","slug":"text-only-domain-adaptation-for-end-to-end","title":"Text-only domain adaptation for end-to-end ASR using integrated text-to-mel-spectrogram generator","date":"2023-02-27","arxiv_id":"2302.14036","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-ensemble-architecture-for","slug":"efficient-ensemble-architecture-for","title":"Efficient Ensemble for Multimodal Punctuation Restoration using Time-Delay Neural Network","date":"2023-02-26","arxiv_id":"2302.13376","repositories_listed":1,"syntology":null},{"url":"/paper/improving-massively-multilingual-asr-with","slug":"improving-massively-multilingual-asr-with","title":"Improving Massively Multilingual ASR With Auxiliary CTC Objectives","date":"2023-02-24","arxiv_id":"2302.12829","repositories_listed":1,"syntology":null},{"url":"/paper/pre-finetuning-for-few-shot-emotional-speech","slug":"pre-finetuning-for-few-shot-emotional-speech","title":"Pre-Finetuning for Few-Shot Emotional Speech Recognition","date":"2023-02-24","arxiv_id":"2302.12921","repositories_listed":1,"syntology":null},{"url":"/paper/gradient-remedy-for-multi-task-learning-in","slug":"gradient-remedy-for-multi-task-learning-in","title":"Gradient Remedy for Multi-Task Learning in End-to-End Noise-Robust Speech Recognition","date":"2023-02-22","arxiv_id":"2302.11362","repositories_listed":1,"syntology":null},{"url":"/paper/a-sidecar-separator-can-convert-a-single","slug":"a-sidecar-separator-can-convert-a-single","title":"A Sidecar Separator Can Convert a Single-Talker Speech Recognition System to a Multi-Talker One","date":"2023-02-20","arxiv_id":"2302.09908","repositories_listed":1,"syntology":null},{"url":"/paper/confidence-score-based-speaker-adaptation-of","slug":"confidence-score-based-speaker-adaptation-of","title":"Confidence Score Based Speaker Adaptation of Conformer Speech Recognition Systems","date":"2023-02-15","arxiv_id":"2302.07521","repositories_listed":1,"syntology":null},{"url":"/paper/readin-a-chinese-multi-task-benchmark-with","slug":"readin-a-chinese-multi-task-benchmark-with","title":"READIN: A Chinese Multi-Task Benchmark with Realistic and Diverse Input Noises","date":"2023-02-14","arxiv_id":"2302.07324","repositories_listed":1,"syntology":null},{"url":"/paper/sneaky-spikes-uncovering-stealthy-backdoor","slug":"sneaky-spikes-uncovering-stealthy-backdoor","title":"Sneaky Spikes: Uncovering Stealthy Backdoor Attacks in Spiking Neural Networks with Neuromorphic Data","date":"2023-02-13","arxiv_id":"2302.06279","repositories_listed":1,"syntology":null},{"url":"/paper/asdf-a-differential-testing-framework-for","slug":"asdf-a-differential-testing-framework-for","title":"ASDF: A Differential Testing Framework for Automatic Speech Recognition Systems","date":"2023-02-11","arxiv_id":"2302.05582","repositories_listed":1,"syntology":null},{"url":"/paper/complex-dynamic-neurons-improved-spiking","slug":"complex-dynamic-neurons-improved-spiking","title":"Complex Dynamic Neurons Improved Spiking Transformer Network for Efficient Automatic Speech Recognition","date":"2023-02-02","arxiv_id":"2302.01194","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/complex-dynamic-neurons-improved-spiking#ran","syntology_url":"https://syntology.ai/paper/2302.01194","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.01194"}},"official":{"repos":["MingLunHan/CIF-PyTorch"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-architecture-search-insights-from-1000","slug":"neural-architecture-search-insights-from-1000","title":"Neural Architecture Search: Insights from 1000 Papers","date":"2023-01-20","arxiv_id":"2301.08727","repositories_listed":1,"syntology":null},{"url":"/paper/syllable-subword-tokens-for-open-vocabulary","slug":"syllable-subword-tokens-for-open-vocabulary","title":"Syllable Subword Tokens for Open Vocabulary Speech Recognition in Malayalam","date":"2023-01-17","arxiv_id":"2301.06736","repositories_listed":1,"syntology":null},{"url":"/paper/olkavs-an-open-large-scale-korean-audio","slug":"olkavs-an-open-large-scale-korean-audio","title":"OLKAVS: An Open Large-Scale Korean Audio-Visual Speech Dataset","date":"2023-01-16","arxiv_id":"2301.06375","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/olkavs-an-open-large-scale-korean-audio#ran","syntology_url":"https://syntology.ai/paper/2301.06375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.06375"}},"official":{"repos":["iip-sogang/olkavs-avspeech"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/audio-visual-efficient-conformer-for-robust","slug":"audio-visual-efficient-conformer-for-robust","title":"Audio-Visual Efficient Conformer for Robust Speech Recognition","date":"2023-01-04","arxiv_id":"2301.01456","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/audio-visual-efficient-conformer-for-robust#ran","syntology_url":"https://syntology.ai/paper/2301.01456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.01456"}},"official":{"repos":["burchim/avec"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/supervised-acoustic-embeddings-and-their","slug":"supervised-acoustic-embeddings-and-their","title":"Supervised Acoustic Embeddings And Their Transferability Across Languages","date":"2023-01-03","arxiv_id":"2301.01020","repositories_listed":1,"syntology":null},{"url":"/paper/towards-voice-reconstruction-from-eeg-during","slug":"towards-voice-reconstruction-from-eeg-during","title":"Towards Voice Reconstruction from EEG during Imagined Speech","date":"2023-01-02","arxiv_id":"2301.07173","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-voice-reconstruction-from-eeg-during#ran","syntology_url":"https://syntology.ai/paper/2301.07173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07173"}},"official":{"repos":["youngeun1209/neurotalk"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-detect-noisy-labels-using-model","slug":"learning-to-detect-noisy-labels-using-model","title":"Learning to Detect Noisy Labels Using Model-Based Features","date":"2022-12-28","arxiv_id":"2212.13767","repositories_listed":1,"syntology":null},{"url":"/paper/skit-s2i-an-indian-accented-speech-to-intent","slug":"skit-s2i-an-indian-accented-speech-to-intent","title":"Skit-S2I: An Indian Accented Speech to Intent dataset","date":"2022-12-26","arxiv_id":"2212.13015","repositories_listed":1,"syntology":null},{"url":"/paper/effectiveness-of-text-acoustic-and-lattice","slug":"effectiveness-of-text-acoustic-and-lattice","title":"Effectiveness of Text, Acoustic, and Lattice-based representations in Spoken Language Understanding tasks","date":"2022-12-16","arxiv_id":"2212.08489","repositories_listed":1,"syntology":null},{"url":"/paper/jointly-learning-visual-and-auditory-speech","slug":"jointly-learning-visual-and-auditory-speech","title":"Jointly Learning Visual and Auditory Speech Representations from Raw Data","date":"2022-12-12","arxiv_id":"2212.06246","repositories_listed":1,"syntology":null},{"url":"/paper/baspro-a-balanced-script-producer-for-speech","slug":"baspro-a-balanced-script-producer-for-speech","title":"BASPRO: a balanced script producer for speech corpus collection based on the genetic algorithm","date":"2022-12-11","arxiv_id":"2301.04120","repositories_listed":1,"syntology":null},{"url":"/paper/lmec-learnable-multiplicative-absolute","slug":"lmec-learnable-multiplicative-absolute","title":"LMEC: Learnable Multiplicative Absolute Position Embedding Based Conformer for Speech Recognition","date":"2022-12-05","arxiv_id":"2212.02099","repositories_listed":1,"syntology":null},{"url":"/paper/softctc-unicode-x2013-semi-supervised","slug":"softctc-unicode-x2013-semi-supervised","title":"SoftCTC -- Semi-Supervised Learning for Text Recognition using Soft Pseudo-Labels","date":"2022-12-05","arxiv_id":"2212.02135","repositories_listed":1,"syntology":null},{"url":"/paper/softcorrect-error-correction-with-soft","slug":"softcorrect-error-correction-with-soft","title":"SoftCorrect: Error Correction with Soft Detection for Automatic Speech Recognition","date":"2022-12-02","arxiv_id":"2212.01039","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/softcorrect-error-correction-with-soft#ran","syntology_url":"https://syntology.ai/paper/2212.01039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.01039"}},"official":{"repos":["microsoft/NeuralSpeech"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/euro-espnet-unsupervised-asr-open-source","slug":"euro-espnet-unsupervised-asr-open-source","title":"EURO: ESPnet Unsupervised ASR Open-source Toolkit","date":"2022-11-30","arxiv_id":"2211.17196","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/euro-espnet-unsupervised-asr-open-source#ran","syntology_url":"https://syntology.ai/paper/2211.17196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.17196"}},"official":{"repos":["espnet/espnet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videodubber-machine-translation-with-speech","slug":"videodubber-machine-translation-with-speech","title":"VideoDubber: Machine Translation with Speech-Aware Length Control for Video Dubbing","date":"2022-11-30","arxiv_id":"2211.16934","repositories_listed":1,"syntology":null},{"url":"/paper/mmspeech-multi-modal-multi-task-encoder","slug":"mmspeech-multi-modal-multi-task-encoder","title":"MMSpeech: Multi-modal Multi-task Encoder-Decoder Pre-training for Speech Recognition","date":"2022-11-29","arxiv_id":"2212.00500","repositories_listed":1,"syntology":null},{"url":"/paper/on-word-error-rate-definitions-and-their","slug":"on-word-error-rate-definitions-and-their","title":"On Word Error Rate Definitions and their Efficient Computation for Multi-Speaker Speech Recognition Systems","date":"2022-11-29","arxiv_id":"2211.16112","repositories_listed":1,"syntology":null},{"url":"/paper/mask-the-correct-tokens-an-embarrassingly","slug":"mask-the-correct-tokens-an-embarrassingly","title":"Mask the Correct Tokens: An Embarrassingly Simple Approach for Error Correction","date":"2022-11-23","arxiv_id":"2211.13252","repositories_listed":1,"syntology":null},{"url":"/paper/whose-emotion-matters-speaker-detection","slug":"whose-emotion-matters-speaker-detection","title":"Whose Emotion Matters? Speaking Activity Localisation without Prior Knowledge","date":"2022-11-23","arxiv_id":"2211.15377","repositories_listed":1,"syntology":null},{"url":"/paper/melhubert-a-simplified-hubert-on-mel","slug":"melhubert-a-simplified-hubert-on-mel","title":"MelHuBERT: A simplified HuBERT on Mel spectrograms","date":"2022-11-17","arxiv_id":"2211.09944","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-joint-speech-recognition-and","slug":"streaming-joint-speech-recognition-and","title":"Streaming Joint Speech Recognition and Disfluency Detection","date":"2022-11-16","arxiv_id":"2211.08726","repositories_listed":1,"syntology":null},{"url":"/paper/improving-children-s-speech-recognition-by","slug":"improving-children-s-speech-recognition-by","title":"Improving Children's Speech Recognition by Fine-tuning Self-supervised Adult Speech Representations","date":"2022-11-14","arxiv_id":"2211.07769","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-unified-conformer-structure-from","slug":"towards-a-unified-conformer-structure-from","title":"Towards A Unified Conformer Structure: from ASR to ASV Task","date":"2022-11-14","arxiv_id":"2211.07201","repositories_listed":1,"syntology":null},{"url":"/paper/the-far-side-of-failure-investigating-the","slug":"the-far-side-of-failure-investigating-the","title":"The Far Side of Failure: Investigating the Impact of Speech Recognition Errors on Subsequent Dementia Classification","date":"2022-11-11","arxiv_id":"2211.07430","repositories_listed":1,"syntology":null},{"url":"/paper/comparative-layer-wise-analysis-of-self","slug":"comparative-layer-wise-analysis-of-self","title":"Comparative layer-wise analysis of self-supervised speech models","date":"2022-11-08","arxiv_id":"2211.03929","repositories_listed":1,"syntology":null},{"url":"/paper/robust-unstructured-knowledge-access-in","slug":"robust-unstructured-knowledge-access-in","title":"Robust Unstructured Knowledge Access in Conversational Dialogue with ASR Errors","date":"2022-11-08","arxiv_id":"2211.03990","repositories_listed":1,"syntology":null},{"url":"/paper/towards-improved-room-impulse-response","slug":"towards-improved-room-impulse-response","title":"Towards Improved Room Impulse Response Estimation for Speech Recognition","date":"2022-11-08","arxiv_id":"2211.04473","repositories_listed":1,"syntology":null},{"url":"/paper/intermpl-momentum-pseudo-labeling-with","slug":"intermpl-momentum-pseudo-labeling-with","title":"InterMPL: Momentum Pseudo-Labeling with Intermediate CTC Loss","date":"2022-11-02","arxiv_id":"2211.00795","repositories_listed":1,"syntology":null},{"url":"/paper/losses-can-be-blessings-routing-self","slug":"losses-can-be-blessings-routing-self","title":"Losses Can Be Blessings: Routing Self-Supervised Speech Representations Towards Efficient Multilingual and Multitask Speech Processing","date":"2022-11-02","arxiv_id":"2211.01522","repositories_listed":1,"syntology":null},{"url":"/paper/speech-text-based-multi-modal-training-with","slug":"speech-text-based-multi-modal-training-with","title":"Speech-text based multi-modal training with bidirectional attention for improved speech recognition","date":"2022-11-01","arxiv_id":"2211.00325","repositories_listed":1,"syntology":null},{"url":"/paper/blank-collapse-compressing-ctc-emission-for","slug":"blank-collapse-compressing-ctc-emission-for","title":"Blank Collapse: Compressing CTC emission for the faster decoding","date":"2022-10-31","arxiv_id":"2210.17017","repositories_listed":1,"syntology":null},{"url":"/paper/delay-penalized-transducer-for-low-latency","slug":"delay-penalized-transducer-for-low-latency","title":"Delay-penalized transducer for low-latency streaming ASR","date":"2022-10-31","arxiv_id":"2211.00490","repositories_listed":1,"syntology":null},{"url":"/paper/fast-and-parallel-decoding-for-transducer","slug":"fast-and-parallel-decoding-for-transducer","title":"Fast and parallel decoding for transducer","date":"2022-10-31","arxiv_id":"2211.00484","repositories_listed":1,"syntology":null},{"url":"/paper/predicting-multi-codebook-vector-quantization","slug":"predicting-multi-codebook-vector-quantization","title":"Predicting Multi-Codebook Vector Quantization Indexes for Knowledge Distillation","date":"2022-10-31","arxiv_id":"2211.00508","repositories_listed":1,"syntology":null},{"url":"/paper/improved-acoustic-to-articulatory-inversion","slug":"improved-acoustic-to-articulatory-inversion","title":"Improved acoustic-to-articulatory inversion using representations from pretrained self-supervised learning models","date":"2022-10-30","arxiv_id":"2210.16871","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-spoken-language-understanding-with","slug":"end-to-end-spoken-language-understanding-with","title":"End-to-end Spoken Language Understanding with Tree-constrained Pointer Generator","date":"2022-10-29","arxiv_id":"2210.16554","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-severity-assessment-of-dysarthric","slug":"automatic-severity-assessment-of-dysarthric","title":"Automatic Severity Classification of Dysarthric speech by using Self-supervised Model with Multi-task Learning","date":"2022-10-27","arxiv_id":"2210.15387","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-context-invariance-in-unsupervised","slug":"evaluating-context-invariance-in-unsupervised","title":"Evaluating context-invariance in unsupervised speech representations","date":"2022-10-27","arxiv_id":"2210.15775","repositories_listed":1,"syntology":null}],"record_sha256":"f25f65cfcb3c1db25c8c7e2360bc49661679eabf56b130f1da600fa940179b5c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}