{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/video-recap-recursive-captioning-of-hour-long","title":"Video ReCap: Recursive Captioning of Hour-Long Videos","arxiv_id":"2402.13250","date":"2024-02-20","proceeding":"CVPR 2024 1","authors":["Md Mohaiminul Islam","Ngan Ho","Xitong Yang","Tushar Nagarajan","Lorenzo Torresani","Gedas Bertasius"],"abstract":"Most video captioning models are designed to process short video clips of few seconds and output text describing low-level visual concepts (e.g., objects, scenes, atomic actions). However, most real-world videos last for minutes or hours and have a complex hierarchical structure spanning different temporal granularities. We propose Video ReCap, a recursive video captioning model that can process video inputs of dramatically different lengths (from 1 second to 2 hours) and output video captions at multiple hierarchy levels. The recursive video-language architecture exploits the synergy between different video hierarchies and can process hour-long videos efficiently. We utilize a curriculum learning training scheme to learn the hierarchical structure of videos, starting from clip-level captions describing atomic actions, then focusing on segment-level descriptions, and concluding with generating summaries for hour-long videos. Furthermore, we introduce Ego4D-HCap dataset by augmenting Ego4D with 8,267 manually collected long-range video summaries. Our recursive model can flexibly generate captions at different hierarchy levels while also being useful for other complex video understanding tasks, such as VideoQA on EgoSchema. Data, code, and models are available at: https://sites.google.com/view/vidrecap","url_abs":"https://arxiv.org/abs/2402.13250v6","url_pdf":"https://arxiv.org/pdf/2402.13250v6.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"video-recap-recursive-captioning-of-hour-long","repo_url":"https://github.com/md-mohaiminul/VideoRecap","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"MIT"}},{"paper_slug":"video-recap-recursive-captioning-of-hour-long","repo_url":"https://github.com/tanveer81/rgnet","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok"}}],"tasks":[{"task_slug":null,"task_name":"EgoSchema"},{"task_slug":"video-captioning","task_name":"Video Captioning"},{"task_slug":"video-understanding","task_name":"Video Understanding"},{"task_slug":"zeroshot-video-question-answer","task_name":"Zero-Shot Video Question Answer"}],"methods":[],"datasets_introduced":[{"slug":"ego4d-hcap","name":"Ego4D-HCap","full_name":""}],"methods_introduced":[],"results":[{"leaderboard":"/sota/zero-shot-video-question-answer-on-egoschema-1","task":"Zero-Shot Video Question Answer","dataset":"EgoSchema (fullset)","model":"Video ReCap","rank_in_archive_order":18,"of":29,"metrics":{"Accuracy":"50.23"},"uses_additional_data":false}],"syntology":{"syntology_url":null,"atlas_url":"https://app.syntology.ai/?focus=2402.13250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13250"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/tanveer81/rgnet","reach":{"status":"ok"}},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/md-mohaiminul/VideoRecap","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"ran":3},"by_repo_kind":{"official":{"samples":2,"ran":2,"repositories":1},"listed":{"samples":1,"ran":1,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":1,"samples":[{"code_sha256_prefix":"b2fa65fa4da99cba","entry":"RGNet","repo":"tanveer81/rgnet","repo_kind":"listed","path":"rgnet/model.py","file_url":"https://github.com/tanveer81/rgnet/blob/HEAD/rgnet/model.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"b2fa65fa4da99cba"}},{"code_sha256_prefix":"9173ec5b02744124","entry":"convert_time","repo":"md-mohaiminul/VideoRecap","repo_kind":"official","path":"train_text_only.py","file_url":"https://github.com/md-mohaiminul/VideoRecap/blob/HEAD/train_text_only.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"9173ec5b02744124"}},{"code_sha256_prefix":"f22d84261af86dff","entry":"decode_one","repo":"md-mohaiminul/VideoRecap","repo_kind":"official","path":"eval_run_time.py","file_url":"https://github.com/md-mohaiminul/VideoRecap/blob/HEAD/eval_run_time.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"f22d84261af86dff"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}