{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/speech-to-text/papers/3","list_of":"/task/speech-to-text","task":"Speech-to-Text","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":5,"rows_per_page":100,"rows":[201,300],"of":403,"counts":{"archive_papers_tagged":403,"with_a_code_link":129,"where_syntology_ran_a_sample":24,"not_listed_spam_title":0,"listed":403,"listed_where_code_ran":24,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":18,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":18,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/speech-to-text","prev":"/task/speech-to-text/papers/2","next":"/task/speech-to-text/papers/4","papers":[{"url":null,"slug":"hands-free-vr","title":"Hands-Free VR","date":"2024-02-23","arxiv_id":"2402.15083","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-translation-with-speech-foundation","title":"Speech Translation with Speech Foundation Models and Large Language Models: What is There and What is Missing?","date":"2024-02-19","arxiv_id":"2402.12025","repositories_listed":0,"syntology":null},{"url":null,"slug":"syllable-based-dnn-hmm-cantonese-speech-to-1","title":"Syllable based DNN-HMM Cantonese Speech to Text System","date":"2024-02-13","arxiv_id":"2402.08788","repositories_listed":0,"syntology":null},{"url":null,"slug":"named-entity-recognition-for-address","title":"Named Entity Recognition for Address Extraction in Speech-to-Text Transcriptions Using Synthetic Data","date":"2024-02-08","arxiv_id":"2402.05545","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-case-study-on-filtering-for-end-to-end","title":"A Case Study on Filtering for End-to-End Speech Translation","date":"2024-02-02","arxiv_id":"2402.01945","repositories_listed":0,"syntology":null},{"url":null,"slug":"digits-micro-model-for-accurate-and-secure","title":"Digits micro-model for accurate and secure transactions","date":"2024-02-02","arxiv_id":"2402.01931","repositories_listed":0,"syntology":null},{"url":null,"slug":"prosody-in-cascade-and-direct-speech-to-text","title":"Prosody in Cascade and Direct Speech-to-Text Translation: a case study on Korean Wh-Phrases","date":"2024-02-01","arxiv_id":"2402.00632","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-efficient-personalized-1","title":"Communication-Efficient Personalized Federated Learning for Speech-to-Text Tasks","date":"2024-01-18","arxiv_id":"2401.10070","repositories_listed":0,"syntology":null},{"url":null,"slug":"oava-the-open-audio-visual-archives","title":"OAVA: the open audio-visual archives aggregator","date":"2023-12-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-the-entropy-semiring-for-neural","title":"Revisiting the Entropy Semiring for Neural Speech Recognition","date":"2023-12-13","arxiv_id":"2312.10087","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-monotonic-multihead-attention","title":"Efficient Monotonic Multihead Attention","date":"2023-12-07","arxiv_id":"2312.04515","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speech-to-text-translation-a","title":"End-to-End Speech-to-Text Translation: A Survey","date":"2023-12-02","arxiv_id":"2312.01053","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-teacher-distillation-for-multilingual","title":"Multi-teacher Distillation for Multilingual Spelling Correction","date":"2023-11-20","arxiv_id":"2311.11518","repositories_listed":0,"syntology":null},{"url":null,"slug":"cosmic-data-efficient-instruction-tuning-for","title":"COSMIC: Data Efficient Instruction-tuning For Speech In-Context Learning","date":"2023-11-03","arxiv_id":"2311.02248","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-joint-language-modeling-for-speech","title":"Toward Joint Language Modeling for Speech Units and Text","date":"2023-10-12","arxiv_id":"2310.08715","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-stability-in-simultaneous-speech","title":"Improving Stability in Simultaneous Speech Translation: A Revision-Controllable Decoding Approach","date":"2023-10-06","arxiv_id":"2310.04399","repositories_listed":0,"syntology":null},{"url":null,"slug":"modular-speech-to-text-translation-for-zero","title":"Modular Speech-to-Text Translation for Zero-Shot Cross-Modal Transfer","date":"2023-10-05","arxiv_id":"2310.03724","repositories_listed":0,"syntology":null},{"url":null,"slug":"afrispeech-200-pan-african-accented-speech","title":"AfriSpeech-200: Pan-African Accented Speech Dataset for Clinical and General Domain ASR","date":"2023-09-30","arxiv_id":"2310.00274","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-multi-tasking-for-speech-to-text","title":"Cross-Modal Multi-Tasking for Speech-to-Text Translation via Hard Parameter Sharing","date":"2023-09-27","arxiv_id":"2309.15826","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-automatic-verbatim-transcripts-for","title":"Developing automatic verbatim transcripts for international multilingual meetings: an end-to-end solution","date":"2023-09-27","arxiv_id":"2309.15609","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepfake-audio-as-a-data-augmentation","title":"Deepfake audio as a data augmentation technique for training automatic speech to text transcription models","date":"2023-09-22","arxiv_id":"2309.12802","repositories_listed":0,"syntology":null},{"url":null,"slug":"speechalign-a-framework-for-speech","title":"SpeechAlign: a Framework for Speech Translation Alignment Evaluation","date":"2023-09-20","arxiv_id":"2309.11585","repositories_listed":0,"syntology":null},{"url":null,"slug":"colld-contrastive-layer-to-layer-distillation","title":"CoLLD: Contrastive Layer-to-layer Distillation for Compressing Multilingual Pre-trained Speech Encoders","date":"2023-09-14","arxiv_id":"2309.07707","repositories_listed":0,"syntology":null},{"url":null,"slug":"phantomsound-black-box-query-efficient-audio","title":"PhantomSound: Black-Box, Query-Efficient Audio Adversarial Attack via Split-Second Phoneme Injection","date":"2023-09-13","arxiv_id":"2309.06960","repositories_listed":0,"syntology":null},{"url":null,"slug":"n-gram-boosting-improving-contextual-biasing","title":"N-gram Boosting: Improving Contextual Biasing with Normalized N-gram Targets","date":"2023-08-04","arxiv_id":"2308.02092","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-rnn-transducers-with-acoustic","title":"Improving RNN-Transducers with Acoustic LookAhead","date":"2023-07-11","arxiv_id":"2307.05006","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-decoder-only-architecture-for-speech-to","title":"On decoder-only architecture for speech-to-text and large language model integration","date":"2023-07-08","arxiv_id":"2307.03917","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-comparison-of-pre-trained-models","title":"Performance Comparison of Pre-trained Models for Speech-to-Text in Turkish: Whisper-Small and Wav2Vec2-XLS-R-300M","date":"2023-07-06","arxiv_id":"2307.04765","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-hybrid-ctc-attention-end-to-end","title":"Online Hybrid CTC/Attention End-to-End Automatic Speech Recognition Architecture","date":"2023-07-05","arxiv_id":"2307.02351","repositories_listed":0,"syntology":null},{"url":null,"slug":"audiopalm-a-large-language-model-that-can","title":"AudioPaLM: A Large Language Model That Can Speak and Listen","date":"2023-06-22","arxiv_id":"2306.12925","repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-advances-in-direct-speech-to-text","title":"Recent Advances in Direct Speech-to-text Translation","date":"2023-06-20","arxiv_id":"2306.11646","repositories_listed":0,"syntology":null},{"url":null,"slug":"computational-language-assessment-open-brain","title":"Open Brain AI. Automatic Language Assessment","date":"2023-06-11","arxiv_id":"2306.06693","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-to-text-adapter-and-speech-to-entity","title":"Speech-to-Text Adapter and Speech-to-Entity Retriever Augmented LLMs for Speech Understanding","date":"2023-06-08","arxiv_id":"2306.07944","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-cross-lingual-transfer-learning-for","title":"Improved Cross-Lingual Transfer Learning For Automatic Speech Translation","date":"2023-06-01","arxiv_id":"2306.00789","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategies-for-improving-low-resource-speech","title":"Strategies for improving low resource speech to text translation relying on pre-trained ASR models","date":"2023-05-31","arxiv_id":"2306.00208","repositories_listed":0,"syntology":null},{"url":null,"slug":"stt4sg-350-a-speech-corpus-for-all-swiss","title":"STT4SG-350: A Speech Corpus for All Swiss German Dialect Regions","date":"2023-05-30","arxiv_id":"2305.18855","repositories_listed":0,"syntology":null},{"url":null,"slug":"cif-pt-bridging-speech-and-text","title":"CIF-PT: Bridging Speech and Text Representations for Spoken Language Understanding via Continuous Integrate-and-Fire Pre-Training","date":"2023-05-27","arxiv_id":"2305.17499","repositories_listed":0,"syntology":null},{"url":null,"slug":"viola-unified-codec-language-models-for","title":"VioLA: Unified Codec Language Models for Speech Recognition, Synthesis, and Translation","date":"2023-05-25","arxiv_id":"2305.16107","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-metrics-for-speech-translation","title":"Improving Metrics for Speech Translation","date":"2023-05-22","arxiv_id":"2305.12918","repositories_listed":0,"syntology":null},{"url":null,"slug":"application-agnostic-language-modeling-for-on","title":"Application-Agnostic Language Modeling for On-Device ASR","date":"2023-05-16","arxiv_id":"2305.09764","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-transducer-and-attention-based-encoder","title":"Hybrid Transducer and Attention based Encoder-Decoder Modeling for Speech-to-Text Tasks","date":"2023-05-04","arxiv_id":"2305.03101","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-autoregressive-nlp-tasks-via","title":"Improving Autoregressive NLP Tasks via Modular Linearized Attention","date":"2023-04-17","arxiv_id":"2304.08453","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-speech-to-speech-translation-with","title":"Enhancing Speech-to-Speech Translation with Multiple TTS Targets","date":"2023-04-10","arxiv_id":"2304.04618","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-language-robot-programming-nlp","title":"Natural Language Robot Programming: NLP integrated with autonomous robotic grasping","date":"2023-04-06","arxiv_id":"2304.02993","repositories_listed":0,"syntology":null},{"url":"/paper/improving-the-previous-state-of-the-art","slug":"improving-the-previous-state-of-the-art","title":"Improving the previous state-of-the-art Frisian ASR by fine-tuning XLS-R","date":"2023-03-31","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"wav2vec-and-its-current-potential-to","title":"wav2vec and its current potential to Automatic Speech Recognition in German for the usage in Digital History: A comparative assessment of available ASR-technologies for the use in cultural heritage contexts","date":"2023-03-06","arxiv_id":"2303.06026","repositories_listed":0,"syntology":null},{"url":null,"slug":"google-usm-scaling-automatic-speech","title":"Google USM: Scaling Automatic Speech Recognition Beyond 100 Languages","date":"2023-03-02","arxiv_id":"2303.01037","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-medical-speech-to-text-accuracy","title":"Improving Medical Speech-to-Text Accuracy with Vision-Language Pre-training Model","date":"2023-02-27","arxiv_id":"2303.00091","repositories_listed":0,"syntology":null},{"url":null,"slug":"patcorrect-non-autoregressive-phoneme","title":"PATCorrect: Non-autoregressive Phoneme-augmented Transformer for ASR Error Correction","date":"2023-02-10","arxiv_id":"2302.05040","repositories_listed":0,"syntology":null},{"url":null,"slug":"characterizing-financial-market-coverage","title":"Characterizing Financial Market Coverage using Artificial Intelligence","date":"2023-02-07","arxiv_id":"2302.03694","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-external-off-policy-speech-to-text","title":"Using External Off-Policy Speech-To-Text Mappings in Contextual End-To-End Automated Speech Recognition","date":"2023-01-06","arxiv_id":"2301.02736","repositories_listed":0,"syntology":null},{"url":null,"slug":"pushing-the-performances-of-asr-models-on","title":"Pushing the performances of ASR models on English and Spanish accents","date":"2022-12-22","arxiv_id":"2212.12048","repositories_listed":0,"syntology":null},{"url":null,"slug":"m3st-mix-at-three-levels-for-speech","title":"M3ST: Mix at Three Levels for Speech Translation","date":"2022-12-07","arxiv_id":"2212.03657","repositories_listed":0,"syntology":null},{"url":null,"slug":"handling-and-extracting-key-entities-from","title":"Handling and extracting key entities from customer conversations using Speech recognition and Named Entity recognition","date":"2022-11-28","arxiv_id":"2211.17107","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-speech-emotion-recognition-with","title":"Multilingual Speech Emotion Recognition With Multi-Gating Mechanism and Neural Architecture Search","date":"2022-10-31","arxiv_id":"2211.08237","repositories_listed":0,"syntology":null},{"url":null,"slug":"phonemic-representation-and-transcription-for","title":"Phonemic Representation and Transcription for Speech to Text Applications for Under-resourced Indigenous African Languages: The Case of Kiswahili","date":"2022-10-29","arxiv_id":"2210.16537","repositories_listed":0,"syntology":null},{"url":null,"slug":"named-entity-detection-and-injection-for","title":"Named Entity Detection and Injection for Direct Speech Translation","date":"2022-10-21","arxiv_id":"2210.11981","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-semi-supervised-end-to-end","title":"Improving Semi-supervised End-to-end Automatic Speech Recognition using CycleGAN and Inter-domain Losses","date":"2022-10-20","arxiv_id":"2210.11642","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-and-effective-unsupervised-speech-1","title":"Simple and Effective Unsupervised Speech Translation","date":"2022-10-18","arxiv_id":"2210.10191","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctc-alignments-improve-autoregressive","title":"CTC Alignments Improve Autoregressive Translation","date":"2022-10-11","arxiv_id":"2210.05200","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-to-text-and-evaluation-of-multiple","title":"Speech-to-Text and Evaluation of Multiple Machine Translation Systems","date":"2022-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"kencorpus-a-kenyan-language-corpus-of-swahili","title":"Kencorpus: A Kenyan Language Corpus of Swahili, Dholuo and Luhya for Natural Language Processing Tasks","date":"2022-08-25","arxiv_id":"2208.12081","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-hypernasality-estimation-with","title":"Improving Hypernasality Estimation with Automatic Speech Recognition in Cleft Palate Speech","date":"2022-08-10","arxiv_id":"2208.05122","repositories_listed":0,"syntology":null},{"url":null,"slug":"extending-rnn-t-based-speech-recognition","title":"Extending RNN-T-based speech recognition systems with emotion and language classification","date":"2022-07-28","arxiv_id":"2207.13965","repositories_listed":0,"syntology":null},{"url":null,"slug":"rsd-gan-regularized-sobolev-defense-gan","title":"RSD-GAN: Regularized Sobolev Defense GAN Against Speech-to-Text Adversarial Attacks","date":"2022-07-14","arxiv_id":"2207.06858","repositories_listed":0,"syntology":null},{"url":null,"slug":"findings-of-the-third-workshop-on-automatic","title":"Findings of the Third Workshop on Automatic Simultaneous Translation","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"language-model-augmented-monotonic-attention","title":"Language Model Augmented Monotonic Attention for Simultaneous Translation","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"swiss-german-speech-to-text-system-evaluation","title":"Swiss German Speech to Text system evaluation","date":"2022-07-01","arxiv_id":"2207.00412","repositories_listed":0,"syntology":null},{"url":null,"slug":"system-description-on-automatic-simultaneous-1","title":"System Description on Automatic Simultaneous Translation Workshop","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-a-speech-recognition-system-for","title":"Developing a Speech Recognition System for Recognizing Tonal Speech Signals Using a Convolutional Neural Network","date":"2022-06-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-semi-automated-live-interlingual","title":"A Semi-Automated Live Interlingual Communication Workflow Featuring Intralingual Respeaking: Evaluation and Benchmarking","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-nos-project-opening-routes-for-the","title":"The Nós Project: Opening routes for the Galician language in the field of language technologies","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-large-vocabulary-kazakh-russian-sign","title":"Towards Large Vocabulary Kazakh-Russian Sign Language Dataset: KRSL-OnlineSchool","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"clinical-dialogue-transcription-error","title":"Clinical Dialogue Transcription Error Correction using Seq2Seq Models","date":"2022-05-26","arxiv_id":"2205.13572","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-preserved-communication-system-for","title":"Semantic-preserved Communication System for Highly Efficient Speech Transmission","date":"2022-05-25","arxiv_id":"2205.12727","repositories_listed":0,"syntology":null},{"url":null,"slug":"samu-xlsr-semantically-aligned-multimodal","title":"SAMU-XLSR: Semantically-Aligned Multimodal Utterance-level Cross-Lingual Speech Representation","date":"2022-05-17","arxiv_id":"2205.08180","repositories_listed":0,"syntology":null},{"url":null,"slug":"hearing-voices-at-the-national-library-a","title":"Hearing voices at the National Library -- a speech corpus and acoustic model for the Swedish language","date":"2022-05-06","arxiv_id":"2205.03026","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-of-a-novel-korean-learning-application","title":"Design of a novel Korean learning application for efficient pronunciation correction","date":"2022-05-04","arxiv_id":"2205.02001","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-adaptive-segmentation-policy-for-end","title":"Learning Adaptive Segmentation Policy for End-to-End Simultaneous Translation","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"naist-simultaneous-speech-to-text-translation","title":"NAIST Simultaneous Speech-to-Text Translation System for IWSLT 2022","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-aisp-sjtu-simultaneous-translation-system","title":"The AISP-SJTU Simultaneous Translation System for IWSLT 2022","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-hw-tscs-simultaneous-speech-translation","title":"The HW-TSC’s Simultaneous Speech Translation System for IWSLT 2022 Evaluation","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"wabert-a-low-resource-end-to-end-model-for","title":"WaBERT: A Low-resource End-to-end Model for Spoken Language Understanding and Speech-to-BERT Alignment","date":"2022-04-22","arxiv_id":"2204.10461","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-direct-speech-to-speech-translation","title":"Enhanced Direct Speech-to-Speech Translation Using Self-supervised Pre-training and Data Augmentation","date":"2022-04-06","arxiv_id":"2204.02967","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-gender-impact-in-self-supervised","title":"A Study of Gender Impact in Self-supervised Models for Speech-to-Text Systems","date":"2022-04-04","arxiv_id":"2204.01397","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-speech-based-end-to-end-automated-speech","title":"Deep Speech Based End-to-End Automated Speech Recognition (ASR) for Indian-English Accents","date":"2022-04-03","arxiv_id":"2204.00977","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-mit-voice-name-system","title":"The MIT Voice Name System","date":"2022-03-28","arxiv_id":"2204.09657","repositories_listed":0,"syntology":null},{"url":null,"slug":"xtreme-s-evaluating-cross-lingual-speech","title":"XTREME-S: Evaluating Cross-lingual Speech Representations","date":"2022-03-21","arxiv_id":"2203.10752","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-combined-approach-to-the-analysis-of-speech","title":"A combined approach to the analysis of speech conversations in a contact center domain","date":"2022-03-12","arxiv_id":"2203.06396","repositories_listed":0,"syntology":null},{"url":null,"slug":"attacks-as-defenses-designing-robust-audio","title":"Attacks as Defenses: Designing Robust Audio CAPTCHAs Using Attacks on Automatic Speech Recognition Systems","date":"2022-03-10","arxiv_id":"2203.05408","repositories_listed":0,"syntology":null},{"url":null,"slug":"which-french-speech-recognition-system-for","title":"Which French speech recognition system for assistant robots?","date":"2022-03-04","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"punctuation-restoration-in-swedish-through","title":"Punctuation restoration in Swedish through fine-tuned KB-BERT","date":"2022-02-14","arxiv_id":"2202.06769","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-aware-speech-to-text-transmission","title":"Semantic-aware Speech to Text Transmission with Redundancy Removal","date":"2022-02-07","arxiv_id":"2202.03211","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-of-a-real-time-wavelet-based","title":"Optimization of a Real-Time Wavelet-Based Algorithm for Improving Speech Intelligibility","date":"2022-02-05","arxiv_id":"2202.02545","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-contrastive-learning-for-speech","title":"Cross-modal Contrastive Learning for Speech Translation","date":"2021-12-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"training-end-to-end-speech-to-text-models-on","title":"Training end-to-end speech-to-text models on mobile phones","date":"2021-12-07","arxiv_id":"2112.03871","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experiment-on-speech-to-text-translation","title":"An Experiment on Speech-to-Text Translation Systems for Manipuri to English on Low Resource Setting","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-of-microphone-position-measurement","title":"Impact of Microphone position Measurement Error on Multi Channel Distant Speech Recognition & Intelligibility","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improve-sinhala-speech-recognition-through","title":"Improve Sinhala Speech Recognition Through e2e LF-MMI Model","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-svd-and-factorized-tdnn","title":"Comparison of SVD and factorized TDNN approaches for speech to text","date":"2021-10-13","arxiv_id":"2110.07027","repositories_listed":0,"syntology":null}],"record_sha256":"1245ee66765ba72cc5cf06f8ead83f55e9632c60f49eec4fe937fa9269ad5681","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}