{"url":"/method/videobert","slug":"videobert","name":"VideoBERT","full_name":"VideoBERT","full_name_withheld":false,"description_markdown":"VideoBERT adapts the powerful [BERT](https://paperswithcode.com/method/bert) model to learn a joint visual-linguistic representation for video. It is used in numerous tasks, including action classification and video captioning.","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"https://arxiv.org/abs/1904.01766v2","title":"VideoBERT: A Joint Model for Video and Language Representation Learning","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Transformers","url":"/methods/category/transformers","pwc_aliases":[]}],"n_papers_tagged":3,"archive_num_papers":null,"papers_newest_first":[{"paper":null,"title":"Understanding Chinese Video and Language via Contrastive Multimodal Pre-Training","date":"2021-04-19","arxiv_id":"2104.09411","n_code_links":0,"syntology":null},{"paper":null,"title":"Visual Grounding Strategies for Text-Only Natural Language Processing","date":"2021-03-25","arxiv_id":"2103.13942","n_code_links":0,"syntology":null},{"paper":"/paper/videobert-a-joint-model-for-video-and","title":"VideoBERT: A Joint Model for Video and Language Representation Learning","date":"2019-04-03","arxiv_id":"1904.01766","n_code_links":3,"syntology":{"ran":0,"of":1,"unverified":1,"pointer_only":0}}],"papers_shown":3,"tasks":[{"task":"/task/language-modeling","name":"Language Modeling","papers":3},{"task":"/task/language-modelling","name":"Language Modelling","papers":3},{"task":"/task/action-classification","name":"Action Classification","papers":1},{"task":"/task/contrastive-learning","name":"Contrastive Learning","papers":1},{"task":"/task/classification","name":"General Classification","papers":1},{"task":"/task/image-retrieval","name":"Image Retrieval","papers":1},{"task":"/task/masked-language-modeling","name":"Masked Language Modeling","papers":1},{"task":"/task/quantization","name":"Quantization","papers":1},{"task":"/task/question-answering","name":"Question Answering","papers":1},{"task":"/task/representation-learning","name":"Representation Learning","papers":1},{"task":"/task/retrieval","name":"Retrieval","papers":1},{"task":"/task/self-supervised-learning","name":"Self-Supervised Learning","papers":1},{"task":"/task/sentence","name":"Sentence","papers":1},{"task":"/task/speech-recognition","name":"Speech Recognition","papers":1},{"task":"/task/video-captioning","name":"Video Captioning","papers":1},{"task":"/task/visual-grounding","name":"Visual Grounding","papers":1},{"task":"/task/visual-question-answering-1","name":"Visual Question Answering","papers":1},{"task":"/task/visual-question-answering","name":"Visual Question Answering (VQA)","papers":1},{"task":"/task/speech-recognition-1","name":"speech-recognition","papers":1}],"tasks_shown":19,"n_tasks":19,"usage_by_year":[{"year":"2019","papers":1},{"year":"2021","papers":2}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/videobert"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}