{"url":"/method/covr","slug":"covr","name":"CoVR","full_name":"Composed Video Retrieval","full_name_withheld":false,"description_markdown":"The composed video retrieval (CoVR) task is a new task, where the goal is to find a video that matches both a query image and a query text. The query image represents a visual concept that the user is interested in, and the query text specifies how the concept should be modified or refined. For example, given an image of a fountain and the text _during show at night_, the CoVR task is to retrieve a video that shows the fountain at night with a show.","description_state":"present","introduced_year":null,"introduced_by":{"title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","paper":"/paper/covr-learning-composed-video-retrieval-from","first_author":"Lucas Ventura","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/covr-learning-composed-video-retrieval-from"},"source":{"url":"https://arxiv.org/abs/2308.14746v4","title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Video-Text Retrieval Models","url":"/methods/category/video-text-retrieval-models","pwc_aliases":[]}],"n_papers_tagged":4,"archive_num_papers":4,"papers_newest_first":[{"paper":"/paper/from-play-to-replay-composed-video-retrieval","title":"From Play to Replay: Composed Video Retrieval for Temporally Fine-Grained Videos","date":"2025-06-05","arxiv_id":"2506.05274","n_code_links":1,"syntology":null},{"paper":"/paper/vdebugger-harnessing-execution-feedback-for","title":"VDebugger: Harnessing Execution Feedback for Debugging Visual Programs","date":"2024-06-19","arxiv_id":"2406.13444","n_code_links":1,"syntology":null},{"paper":"/paper/composed-video-retrieval-via-enriched-context","title":"Composed Video Retrieval via Enriched Context and Discriminative Embeddings","date":"2024-03-25","arxiv_id":"2403.16997","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":"/paper/covr-learning-composed-video-retrieval-from","title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","date":"2023-08-28","arxiv_id":"2308.14746","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}}],"papers_shown":4,"tasks":[{"task":"/task/composed-video-retrieval-covr","name":"Composed Video Retrieval (CoVR)","papers":3},{"task":"/task/retrieval","name":"Retrieval","papers":3},{"task":"/task/video-retrieval","name":"Video Retrieval","papers":3},{"task":"/task/action-classification","name":"Action Classification","papers":1},{"task":"/task/composed-image-retrieval","name":"Composed Image Retrieval (CoIR)","papers":1},{"task":"/task/contrastive-learning","name":"Contrastive Learning","papers":1},{"task":"/task/image-retrieval","name":"Image Retrieval","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/large-language-model","name":"Large Language Model","papers":1},{"task":null,"name":"Triplet","papers":1},{"task":"/task/visual-reasoning","name":"Visual Reasoning","papers":1},{"task":"/task/zero-shot-composed-image-retrieval-zs-cir","name":"Zero-Shot Composed Image Retrieval (ZS-CIR)","papers":1}],"tasks_shown":12,"n_tasks":12,"usage_by_year":[{"year":"2023","papers":1},{"year":"2024","papers":2},{"year":"2025","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/covr"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}