{"url":"/method/ctal","slug":"ctal","name":"CTAL","full_name":"CTAL","full_name_withheld":false,"description_markdown":"**CTAL** is a pre-training framework for strong audio-and-language representations with a [Transformer](https://paperswithcode.com/method/transformer), which aims to learn the intra-modality and inter-modalities connections between audio and language through two proxy tasks on a large amount of audio- and-language pairs: masked language modeling and masked cross-modal acoustic modeling. The pre-trained model is a Transformer for Audio and Language, i.e., CTAL, which consists of two modules, a language stream encoding module which adapts word as input element, and a text-referred audio stream encoder module which accepts both frame-level Mel-spectrograms and token-level output embeddings from the language stream","description_state":"present","introduced_year":null,"introduced_by":{"title":"CTAL: Pre-training Cross-modal Transformer for Audio-and-Language Representations","paper":"/paper/ctal-pre-training-cross-modal-transformer-for","first_author":"Hang Li","n_authors":5,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/ctal-pre-training-cross-modal-transformer-for"},"source":{"url":"https://arxiv.org/abs/2109.00181v1","title":"CTAL: Pre-training Cross-modal Transformer for Audio-and-Language Representations","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Multi-Modal Methods","url":"/methods/category/multi-modal-methods","pwc_aliases":[]},{"area":"Audio","area_id":"audio","collection":"Generative Audio Models","url":"/methods/category/generative-audio-models","pwc_aliases":[]}],"n_papers_tagged":2,"archive_num_papers":2,"papers_newest_first":[{"paper":"/paper/ema-net-efficient-multitask-affinity-learning","title":"Cross-Task Affinity Learning for Multitask Dense Scene Predictions","date":"2024-01-20","arxiv_id":"2401.11124","n_code_links":1,"syntology":null},{"paper":"/paper/ctal-pre-training-cross-modal-transformer-for","title":"CTAL: Pre-training Cross-modal Transformer for Audio-and-Language Representations","date":"2021-09-01","arxiv_id":"2109.00181","n_code_links":1,"syntology":{"ran":3,"of":19,"unverified":16,"pointer_only":5}}],"papers_shown":2,"tasks":[{"task":"/task/decoder","name":"Decoder","papers":1},{"task":"/task/emotion-classification","name":"Emotion Classification","papers":1},{"task":"/task/language-modeling","name":"Language Modeling","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/masked-language-modeling","name":"Masked Language Modeling","papers":1},{"task":"/task/sentiment-analysis","name":"Sentiment Analysis","papers":1},{"task":"/task/speaker-verification","name":"Speaker Verification","papers":1}],"tasks_shown":7,"n_tasks":7,"usage_by_year":[{"year":"2021","papers":1},{"year":"2024","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/ctal"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}