{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-recognition-1/papers/6","list_of":"/task/speech-recognition-1","task":"speech-recognition","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":58,"rows_per_page":100,"rows":[501,600],"of":5715,"counts":{"archive_papers_tagged":5715,"with_a_code_link":1277,"where_syntology_ran_a_sample":162,"not_listed_spam_title":0,"listed":5715,"listed_where_code_ran":162,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":134,"every_run_a_failure_of_syntologys_instrument":28,"listed_with_a_run_with_no_instrument_failure":134,"listed_every_run_a_failure_of_syntologys_instrument":28,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-recognition-1","prev":"/task/speech-recognition-1/papers/5","next":"/task/speech-recognition-1/papers/7","papers":[{"url":"/paper/pseudo-labeling-for-domain-agnostic-bangla","slug":"pseudo-labeling-for-domain-agnostic-bangla","title":"Pseudo-Labeling for Domain-Agnostic Bangla Automatic Speech Recognition","date":"2023-11-06","arxiv_id":"2311.03196","repositories_listed":1,"syntology":null},{"url":"/paper/distilwhisper-efficient-distillation-of-multi","slug":"distilwhisper-efficient-distillation-of-multi","title":"Multilingual DistilWhisper: Efficient Distillation of Multi-task Speech Models via Language-Specific Experts","date":"2023-11-02","arxiv_id":"2311.01070","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-single-channel-speaker-turn-aware","slug":"end-to-end-single-channel-speaker-turn-aware","title":"End-to-End Single-Channel Speaker-Turn Aware Conversational Speech Translation","date":"2023-11-01","arxiv_id":"2311.00697","repositories_listed":1,"syntology":null},{"url":"/paper/mixrep-hidden-representation-mixup-for-low","slug":"mixrep-hidden-representation-mixup-for-low","title":"MixRep: Hidden Representation Mixup for Low-Resource Speech Recognition","date":"2023-10-27","arxiv_id":"2310.18450","repositories_listed":1,"syntology":null},{"url":"/paper/torchaudio-2-1-advancing-speech-recognition","slug":"torchaudio-2-1-advancing-speech-recognition","title":"TorchAudio 2.1: Advancing speech recognition, self-supervised learning, and audio processing components for PyTorch","date":"2023-10-27","arxiv_id":"2310.17864","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/torchaudio-2-1-advancing-speech-recognition#ran","syntology_url":"https://syntology.ai/paper/2310.17864","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17864"}},"official":{"repos":["pytorch/audio"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/whisper-mce-whisper-model-finetuned-for","slug":"whisper-mce-whisper-model-finetuned-for","title":"Developing a Multilingual Dataset and Evaluation Metrics for Code-Switching: A Focus on Hong Kong's Polylingual Dynamics","date":"2023-10-27","arxiv_id":"2310.17953","repositories_listed":1,"syntology":null},{"url":"/paper/artst-arabic-text-and-speech-transformer","slug":"artst-arabic-text-and-speech-transformer","title":"ArTST: Arabic Text and Speech Transformer","date":"2023-10-25","arxiv_id":"2310.16621","repositories_listed":1,"syntology":null},{"url":"/paper/cl-masr-a-continual-learning-benchmark-for","slug":"cl-masr-a-continual-learning-benchmark-for","title":"CL-MASR: A Continual Learning Benchmark for Multilingual ASR","date":"2023-10-25","arxiv_id":"2310.16931","repositories_listed":1,"syntology":null},{"url":"/paper/disco-a-large-scale-human-annotated-corpus","slug":"disco-a-large-scale-human-annotated-corpus","title":"DISCO: A Large Scale Human Annotated Corpus for Disfluency Correction in Indo-European Languages","date":"2023-10-25","arxiv_id":"2310.16749","repositories_listed":1,"syntology":null},{"url":"/paper/accented-speech-recognition-with-accent","slug":"accented-speech-recognition-with-accent","title":"Accented Speech Recognition With Accent-specific Codebooks","date":"2023-10-24","arxiv_id":"2310.15970","repositories_listed":1,"syntology":null},{"url":"/paper/how-much-context-does-my-attention-based-asr","slug":"how-much-context-does-my-attention-based-asr","title":"How Much Context Does My Attention-Based ASR System Need?","date":"2023-10-24","arxiv_id":"2310.15672","repositories_listed":1,"syntology":null},{"url":"/paper/delayed-memory-unit-modelling-temporal","slug":"delayed-memory-unit-modelling-temporal","title":"Delayed Memory Unit: Modelling Temporal Dependency Through Delay Gate","date":"2023-10-23","arxiv_id":"2310.14982","repositories_listed":1,"syntology":null},{"url":"/paper/key-frame-mechanism-for-efficient-conformer","slug":"key-frame-mechanism-for-efficient-conformer","title":"Key Frame Mechanism For Efficient Conformer Based End-to-end Speech Recognition","date":"2023-10-23","arxiv_id":"2310.14954","repositories_listed":1,"syntology":null},{"url":"/paper/salmonn-towards-generic-hearing-abilities-for","slug":"salmonn-towards-generic-hearing-abilities-for","title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","date":"2023-10-20","arxiv_id":"2310.13289","repositories_listed":1,"syntology":null},{"url":"/paper/zipformer-a-faster-and-better-encoder-for","slug":"zipformer-a-faster-and-better-encoder-for","title":"Zipformer: A faster and better encoder for automatic speech recognition","date":"2023-10-17","arxiv_id":"2310.11230","repositories_listed":1,"syntology":null},{"url":"/paper/homophone-disambiguation-reveals-patterns-of","slug":"homophone-disambiguation-reveals-patterns-of","title":"Homophone Disambiguation Reveals Patterns of Context Mixing in Speech Transformers","date":"2023-10-15","arxiv_id":"2310.09925","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-test-time-adaptation-for-acoustic","slug":"advancing-test-time-adaptation-for-acoustic","title":"Advancing Test-Time Adaptation in Wild Acoustic Test Settings","date":"2023-10-14","arxiv_id":"2310.09505","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/advancing-test-time-adaptation-for-acoustic#ran","syntology_url":"https://syntology.ai/paper/2310.09505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09505"}},"official":{"repos":["Waffle-Liu/CEA"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/salm-speech-augmented-language-model-with-in","slug":"salm-speech-augmented-language-model-with-in","title":"SALM: Speech-augmented Language Model with In-context Learning for Speech Recognition and Translation","date":"2023-10-13","arxiv_id":"2310.09424","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-the-adapters-for-code-switching-in","slug":"adapting-the-adapters-for-code-switching-in","title":"Adapting the adapters for code-switching in multilingual ASR","date":"2023-10-11","arxiv_id":"2310.07423","repositories_listed":1,"syntology":null},{"url":"/paper/no-pitch-left-behind-addressing-gender","slug":"no-pitch-left-behind-addressing-gender","title":"No Pitch Left Behind: Addressing Gender Unbalance in Automatic Speech Recognition through Pitch Manipulation","date":"2023-10-10","arxiv_id":"2310.06590","repositories_listed":1,"syntology":null},{"url":"/paper/whispering-llama-a-cross-modal-generative","slug":"whispering-llama-a-cross-modal-generative","title":"Whispering LLaMA: A Cross-Modal Generative Error Correction Framework for Speech Recognition","date":"2023-10-10","arxiv_id":"2310.06434","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/whispering-llama-a-cross-modal-generative#ran","syntology_url":"https://syntology.ai/paper/2310.06434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06434"}},"official":{"repos":["srijith-rkr/whispering-llama"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-multilingual-self-supervised","slug":"leveraging-multilingual-self-supervised","title":"Leveraging Multilingual Self-Supervised Pretrained Models for Sequence-to-Sequence End-to-End Spoken Language Understanding","date":"2023-10-09","arxiv_id":"2310.06103","repositories_listed":1,"syntology":null},{"url":"/paper/ed-cec-improving-rare-word-recognition-using","slug":"ed-cec-improving-rare-word-recognition-using","title":"ed-cec: improving rare word recognition using asr postprocessing based on error detection and context-aware error correction","date":"2023-10-08","arxiv_id":"2310.05129","repositories_listed":1,"syntology":null},{"url":"/paper/dementia-assessment-using-mandarin-speech","slug":"dementia-assessment-using-mandarin-speech","title":"Dementia Assessment Using Mandarin Speech with an Attention-based Speech Recognition Encoder","date":"2023-10-06","arxiv_id":"2310.03985","repositories_listed":1,"syntology":null},{"url":"/paper/decoderlens-layerwise-interpretation-of","slug":"decoderlens-layerwise-interpretation-of","title":"DecoderLens: Layerwise Interpretation of Encoder-Decoder Transformers","date":"2023-10-05","arxiv_id":"2310.03686","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-speech-recognition-with-n","slug":"unsupervised-speech-recognition-with-n","title":"Unsupervised Speech Recognition with N-Skipgram and Positional Unigram Matching","date":"2023-10-03","arxiv_id":"2310.02382","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unsupervised-speech-recognition-with-n#ran","syntology_url":"https://syntology.ai/paper/2310.02382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02382"}},"official":{"repos":["lwang114/graphunsupasr"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-speech-synthesis-by-training","slug":"evaluating-speech-synthesis-by-training","title":"Evaluating Speech Synthesis by Training Recognizers on Synthetic Speech","date":"2023-10-01","arxiv_id":"2310.00706","repositories_listed":1,"syntology":null},{"url":"/paper/rtfs-net-recurrent-time-frequency-modelling","slug":"rtfs-net-recurrent-time-frequency-modelling","title":"RTFS-Net: Recurrent Time-Frequency Modelling for Efficient Audio-Visual Speech Separation","date":"2023-09-29","arxiv_id":"2309.17189","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/rtfs-net-recurrent-time-frequency-modelling#ran","syntology_url":"https://syntology.ai/paper/2309.17189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17189"}},"official":{"repos":["spkgyk/RTFS-Net"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/hyporadise-an-open-baseline-for-generative-1","slug":"hyporadise-an-open-baseline-for-generative-1","title":"HyPoradise: An Open Baseline for Generative Speech Recognition with Large Language Models","date":"2023-09-27","arxiv_id":"2309.15701","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hyporadise-an-open-baseline-for-generative-1#ran","syntology_url":"https://syntology.ai/paper/2309.15701","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15701"}},"official":{"repos":["hypotheses-paradise/hypo2trans"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/speech-collage-code-switched-audio-generation","slug":"speech-collage-code-switched-audio-generation","title":"Speech collage: code-switched audio generation by collaging monolingual corpora","date":"2023-09-27","arxiv_id":"2309.15674","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-flawed-data-weakly-supervised","slug":"learning-from-flawed-data-weakly-supervised","title":"Learning from Flawed Data: Weakly Supervised Automatic Speech Recognition","date":"2023-09-26","arxiv_id":"2309.15796","repositories_listed":1,"syntology":null},{"url":"/paper/segmentation-free-streaming-machine","slug":"segmentation-free-streaming-machine","title":"Segmentation-Free Streaming Machine Translation","date":"2023-09-26","arxiv_id":"2309.14823","repositories_listed":1,"syntology":null},{"url":"/paper/updated-corpora-and-benchmarks-for-long-form","slug":"updated-corpora-and-benchmarks-for-long-form","title":"Updated Corpora and Benchmarks for Long-Form Speech Recognition","date":"2023-09-26","arxiv_id":"2309.15013","repositories_listed":1,"syntology":null},{"url":"/paper/fast-hubert-an-efficient-training-framework","slug":"fast-hubert-an-efficient-training-framework","title":"Fast-HuBERT: An Efficient Training Framework for Self-Supervised Speech Representation Learning","date":"2023-09-25","arxiv_id":"2309.13860","repositories_listed":1,"syntology":null},{"url":"/paper/human-transcription-quality-improvement","slug":"human-transcription-quality-improvement","title":"Human Transcription Quality Improvement","date":"2023-09-24","arxiv_id":"2309.14372","repositories_listed":1,"syntology":null},{"url":"/paper/big-model-only-for-hard-audios-sample","slug":"big-model-only-for-hard-audios-sample","title":"Big model only for hard audios: Sample dependent Whisper model selection for efficient inferences","date":"2023-09-22","arxiv_id":"2309.12712","repositories_listed":1,"syntology":null},{"url":"/paper/memory-augmented-conformer-for-improved-end","slug":"memory-augmented-conformer-for-improved-end","title":"Memory-augmented conformer for improved end-to-end long-form ASR","date":"2023-09-22","arxiv_id":"2309.13029","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-gaps-of-both-modality-and","slug":"bridging-the-gaps-of-both-modality-and","title":"Bridging the Gaps of Both Modality and Language: Synchronous Bilingual CTC for Speech Translation and Speech Recognition","date":"2023-09-21","arxiv_id":"2309.12234","repositories_listed":1,"syntology":null},{"url":"/paper/comflp-correlation-measure-based-fast-search","slug":"comflp-correlation-measure-based-fast-search","title":"CoMFLP: Correlation Measure based Fast Search on ASR Layer Pruning","date":"2023-09-21","arxiv_id":"2309.11768","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/comflp-correlation-measure-based-fast-search#ran","syntology_url":"https://syntology.ai/paper/2309.11768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11768"}},"official":{"repos":["louislau1129/comflp"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-tuning-self-supervised-learning-models","slug":"fine-tuning-self-supervised-learning-models","title":"Fine-Tuning Self-Supervised Learning Models for End-to-End Pronunciation Scoring","date":"2023-09-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-the-zero-shot-power-of-instruction","slug":"harnessing-the-zero-shot-power-of-instruction","title":"Harnessing the Zero-Shot Power of Instruction-Tuned Large Language Model in End-to-End Speech Recognition","date":"2023-09-19","arxiv_id":"2309.10524","repositories_listed":1,"syntology":null},{"url":"/paper/hypr-a-comprehensive-study-for-asr-hypothesis","slug":"hypr-a-comprehensive-study-for-asr-hypothesis","title":"HypR: A comprehensive study for ASR hypothesis revising with a reference corpus","date":"2023-09-18","arxiv_id":"2309.09838","repositories_listed":1,"syntology":null},{"url":"/paper/training-dynamic-models-using-early-exits-for","slug":"training-dynamic-models-using-early-exits-for","title":"Training dynamic models using early exits for automatic speech recognition on resource-constrained devices","date":"2023-09-18","arxiv_id":"2309.09546","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-quantised-end-to-end-asr-models-via","slug":"enhancing-quantised-end-to-end-asr-models-via","title":"Enhancing Quantised End-to-End ASR Models via Personalisation","date":"2023-09-17","arxiv_id":"2309.09136","repositories_listed":1,"syntology":null},{"url":"/paper/diacorrect-error-correction-back-end-for","slug":"diacorrect-error-correction-back-end-for","title":"DiaCorrect: Error Correction Back-end For Speaker Diarization","date":"2023-09-15","arxiv_id":"2309.08377","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-based-punctuation-restoration-for","slug":"transformer-based-punctuation-restoration-for","title":"Transformer Based Punctuation Restoration for Turkish","date":"2023-09-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unimodal-aggregation-for-ctc-based-speech","slug":"unimodal-aggregation-for-ctc-based-speech","title":"Unimodal Aggregation for CTC-based Speech Recognition","date":"2023-09-15","arxiv_id":"2309.08150","repositories_listed":1,"syntology":null},{"url":"/paper/visual-speech-recognition-for-low-resource","slug":"visual-speech-recognition-for-low-resource","title":"Visual Speech Recognition for Languages with Limited Labeled Data using Automatic Labels from Whisper","date":"2023-09-15","arxiv_id":"2309.08535","repositories_listed":1,"syntology":null},{"url":"/paper/diarist-streaming-speech-translation-with","slug":"diarist-streaming-speech-translation-with","title":"DiariST: Streaming Speech Translation with Speaker Diarization","date":"2023-09-14","arxiv_id":"2309.08007","repositories_listed":1,"syntology":null},{"url":"/paper/funcodec-a-fundamental-reproducible-and","slug":"funcodec-a-fundamental-reproducible-and","title":"FunCodec: A Fundamental, Reproducible and Integrable Open-source Toolkit for Neural Speech Codec","date":"2023-09-14","arxiv_id":"2309.07405","repositories_listed":1,"syntology":null},{"url":"/paper/towards-universal-speech-discrete-tokens-a","slug":"towards-universal-speech-discrete-tokens-a","title":"Towards Universal Speech Discrete Tokens: A Case Study for ASR and TTS","date":"2023-09-14","arxiv_id":"2309.07377","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-asr-for-resource-constrained-robots","slug":"hybrid-asr-for-resource-constrained-robots","title":"Hybrid ASR for Resource-Constrained Robots: HMM - Deep Learning Fusion","date":"2023-09-11","arxiv_id":"2309.07164","repositories_listed":1,"syntology":null},{"url":"/paper/active-learning-for-classifying-2d-grid-based","slug":"active-learning-for-classifying-2d-grid-based","title":"Active Learning for Classifying 2D Grid-Based Level Completability","date":"2023-09-08","arxiv_id":"2309.04367","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/active-learning-for-classifying-2d-grid-based#ran","syntology_url":"https://syntology.ai/paper/2309.04367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.04367"}},"official":{"repos":["mahsabazzaz/level-completabilty-x-active-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-speech-recognition-and-disfluency-1","slug":"end-to-end-speech-recognition-and-disfluency-1","title":"End-to-End Speech Recognition and Disfluency Removal with Acoustic Language Model Pretraining","date":"2023-09-08","arxiv_id":"2309.04516","repositories_listed":1,"syntology":null},{"url":"/paper/perceptual-and-task-oriented-assessment-of-a","slug":"perceptual-and-task-oriented-assessment-of-a","title":"Perceptual and Task-Oriented Assessment of a Semantic Metric for ASR Evaluation","date":"2023-09-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/blsp-bootstrapping-language-speech-pre-1","slug":"blsp-bootstrapping-language-speech-pre-1","title":"BLSP: Bootstrapping Language-Speech Pre-training via Behavior Alignment of Continuation Writing","date":"2023-09-02","arxiv_id":"2309.00916","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/blsp-bootstrapping-language-speech-pre-1#ran","syntology_url":"https://syntology.ai/paper/2309.00916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.00916"}},"official":{"repos":["cwang621/blsp"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/speech-wikimedia-a-77-language-multilingual","slug":"speech-wikimedia-a-77-language-multilingual","title":"Speech Wikimedia: A 77 Language Multilingual Speech Dataset","date":"2023-08-30","arxiv_id":"2308.15710","repositories_listed":1,"syntology":null},{"url":"/paper/a-small-and-fast-bert-for-chinese-medical","slug":"a-small-and-fast-bert-for-chinese-medical","title":"A Small and Fast BERT for Chinese Medical Punctuation Restoration","date":"2023-08-24","arxiv_id":"2308.12568","repositories_listed":1,"syntology":null},{"url":"/paper/an-effective-transformer-based-contextual","slug":"an-effective-transformer-based-contextual","title":"An Effective Transformer-based Contextual Model and Temporal Gate Pooling for Speaker Identification","date":"2023-08-22","arxiv_id":"2308.11241","repositories_listed":1,"syntology":null},{"url":"/paper/bayes-risk-transducer-transducer-with","slug":"bayes-risk-transducer-transducer-with","title":"Bayes Risk Transducer: Transducer with Controllable Alignment Prediction","date":"2023-08-19","arxiv_id":"2308.10107","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-open-vocabulary-keyword-search-1","slug":"end-to-end-open-vocabulary-keyword-search-1","title":"End-to-End Open Vocabulary Keyword Search With Multilingual Neural Representations","date":"2023-08-15","arxiv_id":"2308.08027","repositories_listed":1,"syntology":null},{"url":"/paper/improving-audio-visual-speech-recognition-by","slug":"improving-audio-visual-speech-recognition-by","title":"Improving Audio-Visual Speech Recognition by Lip-Subword Correlation Based Visual Pre-training and Cross-Modal Fusion Encoder","date":"2023-08-14","arxiv_id":"2308.08488","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-emotion-recognition-with-speech","slug":"integrating-emotion-recognition-with-speech","title":"Integrating Emotion Recognition with Speech Recognition and Speaker Diarisation for Conversations","date":"2023-08-14","arxiv_id":"2308.07145","repositories_listed":1,"syntology":null},{"url":"/paper/omnidatacomposer-a-unified-data-structure-for","slug":"omnidatacomposer-a-unified-data-structure-for","title":"OmniDataComposer: A Unified Data Structure for Multimodal Data Fusion and Infinite Data Generation","date":"2023-08-08","arxiv_id":"2308.04126","repositories_listed":1,"syntology":null},{"url":"/paper/on-monotonic-aggregation-for-open-domain-qa","slug":"on-monotonic-aggregation-for-open-domain-qa","title":"On Monotonic Aggregation for Open-domain QA","date":"2023-08-08","arxiv_id":"2308.04176","repositories_listed":1,"syntology":null},{"url":"/paper/mispronunciation-detection-using-self","slug":"mispronunciation-detection-using-self","title":"Mispronunciation detection using self-supervised speech representations","date":"2023-07-30","arxiv_id":"2307.16324","repositories_listed":1,"syntology":null},{"url":"/paper/iroyinspeech-a-multi-purpose-yoruba-speech","slug":"iroyinspeech-a-multi-purpose-yoruba-speech","title":"ÌròyìnSpeech: A multi-purpose Yorùbá Speech Corpus","date":"2023-07-29","arxiv_id":"2307.16071","repositories_listed":1,"syntology":null},{"url":"/paper/turning-whisper-into-real-time-transcription","slug":"turning-whisper-into-real-time-transcription","title":"Turning Whisper into Real-Time Transcription System","date":"2023-07-27","arxiv_id":"2307.14743","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/turning-whisper-into-real-time-transcription#ran","syntology_url":"https://syntology.ai/paper/2307.14743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.14743"}},"official":{"repos":["ufal/whisper_streaming"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/a-model-for-every-user-and-budget-label-free","slug":"a-model-for-every-user-and-budget-label-free","title":"A Model for Every User and Budget: Label-Free and Personalized Mixed-Precision Quantization","date":"2023-07-24","arxiv_id":"2307.12659","repositories_listed":1,"syntology":null},{"url":"/paper/adaptation-of-whisper-models-to-child-speech","slug":"adaptation-of-whisper-models-to-child-speech","title":"Adaptation of Whisper models to child speech recognition","date":"2023-07-24","arxiv_id":"2307.13008","repositories_listed":1,"syntology":null},{"url":"/paper/code-switched-urdu-asr-for-noisy-telephonic","slug":"code-switched-urdu-asr-for-noisy-telephonic","title":"Code-Switched Urdu ASR for Noisy Telephonic Environment using Data Centric Approach with Hybrid HMM and CNN-TDNN","date":"2023-07-24","arxiv_id":"2307.12759","repositories_listed":1,"syntology":null},{"url":"/paper/a-change-of-heart-improving-speech-emotion","slug":"a-change-of-heart-improving-speech-emotion","title":"A Change of Heart: Improving Speech Emotion Recognition through Speech-to-Text Modality Conversion","date":"2023-07-21","arxiv_id":"2307.11584","repositories_listed":1,"syntology":null},{"url":"/paper/topic-identification-for-spontaneous-speech","slug":"topic-identification-for-spontaneous-speech","title":"Topic Identification For Spontaneous Speech: Enriching Audio Features With Embedded Linguistic Information","date":"2023-07-21","arxiv_id":"2307.11450","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-dive-into-the-disparity-of-word-error","slug":"a-deep-dive-into-the-disparity-of-word-error","title":"A Deep Dive into the Disparity of Word Error Rates Across Thousands of NPTEL MOOC Videos","date":"2023-07-20","arxiv_id":"2307.10587","repositories_listed":1,"syntology":null},{"url":"/paper/oxfordvgg-submission-to-the-ego4d-av","slug":"oxfordvgg-submission-to-the-ego4d-av","title":"OxfordVGG Submission to the EGO4D AV Transcription Challenge","date":"2023-07-18","arxiv_id":"2307.09006","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-domain-sensitive-speech-recognition","slug":"zero-shot-domain-sensitive-speech-recognition","title":"Zero-shot Domain-sensitive Speech Recognition with Prompt-conditioning Fine-tuning","date":"2023-07-18","arxiv_id":"2307.10274","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-large-language-model-with-speech-for","slug":"adapting-large-language-model-with-speech-for","title":"Adapting Large Language Model with Speech for Fully Formatted End-to-End Speech Recognition","date":"2023-07-17","arxiv_id":"2307.08234","repositories_listed":1,"syntology":null},{"url":"/paper/towards-stealthy-backdoor-attacks-against","slug":"towards-stealthy-backdoor-attacks-against","title":"Towards Stealthy Backdoor Attacks against Speech Recognition via Elements of Sound","date":"2023-07-17","arxiv_id":"2307.08208","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-stealthy-backdoor-attacks-against#ran","syntology_url":"https://syntology.ai/paper/2307.08208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08208"}},"official":{"repos":["hanbocai/badspeech_soe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sumformer-a-linear-complexity-alternative-to","slug":"sumformer-a-linear-complexity-alternative-to","title":"SummaryMixing: A Linear-Complexity Alternative to Self-Attention for Speech Recognition and Understanding","date":"2023-07-12","arxiv_id":"2307.07421","repositories_listed":1,"syntology":null},{"url":"/paper/writer-adaptation-for-offline-text","slug":"writer-adaptation-for-offline-text","title":"Writer adaptation for offline text recognition: An exploration of neural network-based methods","date":"2023-07-11","arxiv_id":"2307.15071","repositories_listed":1,"syntology":null},{"url":"/paper/gammatonegram-representation-for-end-to-end","slug":"gammatonegram-representation-for-end-to-end","title":"Gammatonegram Representation for End-to-End Dysarthric Speech Processing Tasks: Speech Recognition, Speaker Identification, and Intelligibility Assessment","date":"2023-07-06","arxiv_id":"2307.03296","repositories_listed":1,"syntology":null},{"url":"/paper/using-joint-training-speaker-encoder-with","slug":"using-joint-training-speaker-encoder-with","title":"Using joint training speaker encoder with consistency loss to achieve cross-lingual voice conversion and expressive voice conversion","date":"2023-07-01","arxiv_id":"2307.00393","repositories_listed":1,"syntology":null},{"url":"/paper/learning-delays-in-spiking-neural-networks","slug":"learning-delays-in-spiking-neural-networks","title":"Learning Delays in Spiking Neural Networks using Dilated Convolutions with Learnable Spacings","date":"2023-06-30","arxiv_id":"2306.17670","repositories_listed":1,"syntology":null},{"url":"/paper/lyricwhiz-robust-multilingual-zero-shot","slug":"lyricwhiz-robust-multilingual-zero-shot","title":"LyricWhiz: Robust Multilingual Zero-shot Lyrics Transcription by Whispering to ChatGPT","date":"2023-06-29","arxiv_id":"2306.17103","repositories_listed":1,"syntology":null},{"url":"/paper/long-term-conversation-analysis-exploring","slug":"long-term-conversation-analysis-exploring","title":"Long-term Conversation Analysis: Exploring Utility and Privacy","date":"2023-06-28","arxiv_id":"2306.16071","repositories_listed":1,"syntology":null},{"url":"/paper/a-reference-less-quality-metric-for-automatic","slug":"a-reference-less-quality-metric-for-automatic","title":"A Reference-less Quality Metric for Automatic Speech Recognition via Contrastive-Learning of a Multi-Language Model with Self-Supervision","date":"2023-06-21","arxiv_id":"2306.13114","repositories_listed":1,"syntology":null},{"url":"/paper/norefer-a-referenceless-quality-metric-for","slug":"norefer-a-referenceless-quality-metric-for","title":"NoRefER: a Referenceless Quality Metric for Automatic Speech Recognition via Semi-Supervised Language Model Fine-Tuning with Contrastive Learning","date":"2023-06-21","arxiv_id":"2306.12577","repositories_listed":1,"syntology":null},{"url":"/paper/hk-legicost-leveraging-non-verbatim","slug":"hk-legicost-leveraging-non-verbatim","title":"HK-LegiCoST: Leveraging Non-Verbatim Transcripts for Speech Translation","date":"2023-06-20","arxiv_id":"2306.11252","repositories_listed":1,"syntology":null},{"url":"/paper/rehearsal-free-online-continual-learning-for","slug":"rehearsal-free-online-continual-learning-for","title":"Rehearsal-Free Online Continual Learning for Automatic Speech Recognition","date":"2023-06-19","arxiv_id":"2306.10860","repositories_listed":1,"syntology":null},{"url":"/paper/duta-vc-a-duration-aware-typical-to-atypical","slug":"duta-vc-a-duration-aware-typical-to-atypical","title":"DuTa-VC: A Duration-aware Typical-to-atypical Voice Conversion Approach with Diffusion Probabilistic Model","date":"2023-06-18","arxiv_id":"2306.10588","repositories_listed":1,"syntology":null},{"url":"/paper/hearing-lips-in-noise-universal-viseme","slug":"hearing-lips-in-noise-universal-viseme","title":"Hearing Lips in Noise: Universal Viseme-Phoneme Mapping and Transfer for Robust Audio-Visual Speech Recognition","date":"2023-06-18","arxiv_id":"2306.10563","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hearing-lips-in-noise-universal-viseme#ran","syntology_url":"https://syntology.ai/paper/2306.10563","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10563"}},"official":{"repos":["yuchen005/univpm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mir-gan-refining-frame-level-modality","slug":"mir-gan-refining-frame-level-modality","title":"MIR-GAN: Refining Frame-Level Modality-Invariant Representations with Adversarial Network for Audio-Visual Speech Recognition","date":"2023-06-18","arxiv_id":"2306.10567","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/mir-gan-refining-frame-level-modality#ran","syntology_url":"https://syntology.ai/paper/2306.10567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10567"}},"official":{"repos":["yuchen005/mir-gan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/surt-2-0-advances-in-transducer-based-multi","slug":"surt-2-0-advances-in-transducer-based-multi","title":"SURT 2.0: Advances in Transducer-based Multi-talker Speech Recognition","date":"2023-06-18","arxiv_id":"2306.10559","repositories_listed":1,"syntology":null},{"url":"/paper/italic-an-italian-intent-classification","slug":"italic-an-italian-intent-classification","title":"ITALIC: An Italian Intent Classification Dataset","date":"2023-06-14","arxiv_id":"2306.08502","repositories_listed":1,"syntology":null},{"url":"/paper/towards-training-bilingual-and-code-switched","slug":"towards-training-bilingual-and-code-switched","title":"Unified model for code-switching speech recognition and language identification based on a concatenated tokenizer","date":"2023-06-14","arxiv_id":"2306.08753","repositories_listed":1,"syntology":null},{"url":"/paper/utilizing-longitudinal-chest-x-rays-and","slug":"utilizing-longitudinal-chest-x-rays-and","title":"Utilizing Longitudinal Chest X-Rays and Reports to Pre-Fill Radiology Reports","date":"2023-06-14","arxiv_id":"2306.08749","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-learning-based-audio-to-lyrics","slug":"contrastive-learning-based-audio-to-lyrics","title":"Contrastive Learning-Based Audio to Lyrics Alignment for Multiple Languages","date":"2023-06-13","arxiv_id":"2306.07744","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-training-for-low-resource","slug":"adversarial-training-for-low-resource","title":"Adversarial Training For Low-Resource Disfluency Correction","date":"2023-06-10","arxiv_id":"2306.06384","repositories_listed":1,"syntology":null},{"url":"/paper/opensr-open-modality-speech-recognition-via","slug":"opensr-open-modality-speech-recognition-via","title":"OpenSR: Open-Modality Speech Recognition via Maintaining Multi-Modality Alignment","date":"2023-06-10","arxiv_id":"2306.06410","repositories_listed":1,"syntology":null},{"url":"/paper/a-theory-of-unsupervised-speech-recognition","slug":"a-theory-of-unsupervised-speech-recognition","title":"A Theory of Unsupervised Speech Recognition","date":"2023-06-09","arxiv_id":"2306.07926","repositories_listed":1,"syntology":null}],"record_sha256":"dec5b59eb65bb121a1c1c258b3c8e719ccc4f7560673e29e7ad0ce8031d301e8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}