{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/dense-video-captioning/papers/ran/1","list_of":"/task/dense-video-captioning","task":"Dense Video Captioning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,16],"of":16,"counts":{"archive_papers_tagged":76,"with_a_code_link":39,"where_syntology_ran_a_sample":16,"not_listed_spam_title":0,"listed":76,"listed_where_code_ran":16,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":13,"every_run_a_failure_of_syntologys_instrument":3,"listed_with_a_run_with_no_instrument_failure":13,"listed_every_run_a_failure_of_syntologys_instrument":3,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/dense-video-captioning/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/implicit-location-caption-alignment-via","slug":"implicit-location-caption-alignment-via","title":"Implicit Location-Caption Alignment via Complementary Masking for Weakly-Supervised Dense Video Captioning","date":"2024-12-17","arxiv_id":"2412.12791","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/implicit-location-caption-alignment-via#ran","syntology_url":"https://syntology.ai/paper/2412.12791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12791"}},"official":{"repos":["ShipingGe/ILCACM"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/longvale-vision-audio-language-event","slug":"longvale-vision-audio-language-event","title":"LongVALE: Vision-Audio-Language-Event Benchmark Towards Time-Aware Omni-Modal Perception of Long Videos","date":"2024-11-29","arxiv_id":"2411.19772","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/longvale-vision-audio-language-event#ran","syntology_url":"https://syntology.ai/paper/2411.19772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19772"}},"official":{"repos":["ttgeng233/LongVALE"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/videollm-knows-when-to-speak-enhancing-time","slug":"videollm-knows-when-to-speak-enhancing-time","title":"VideoLLM Knows When to Speak: Enhancing Time-Sensitive Video Comprehension with Video-Text Duet Interaction Format","date":"2024-11-27","arxiv_id":"2411.17991","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videollm-knows-when-to-speak-enhancing-time#ran","syntology_url":"https://syntology.ai/paper/2411.17991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17991"}},"official":{"repos":["yellow-binary-tree/mmduet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grounded-videollm-sharpening-fine-grained","slug":"grounded-videollm-sharpening-fine-grained","title":"Grounded-VideoLLM: Sharpening Fine-grained Temporal Grounding in Video Large Language Models","date":"2024-10-04","arxiv_id":"2410.03290","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounded-videollm-sharpening-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2410.03290","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03290"}},"official":{"repos":["whb139426/grounded-video-llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videogpt-integrating-image-and-video-encoders","slug":"videogpt-integrating-image-and-video-encoders","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","date":"2024-06-13","arxiv_id":"2406.09418","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videogpt-integrating-image-and-video-encoders#ran","syntology_url":"https://syntology.ai/paper/2406.09418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09418"}},"official":{"repos":["mbzuai-oryx/videogpt-plus"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vtg-llm-integrating-timestamp-knowledge-into","slug":"vtg-llm-integrating-timestamp-knowledge-into","title":"VTG-LLM: Integrating Timestamp Knowledge into Video LLMs for Enhanced Video Temporal Grounding","date":"2024-05-22","arxiv_id":"2405.13382","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vtg-llm-integrating-timestamp-knowledge-into#ran","syntology_url":"https://syntology.ai/paper/2405.13382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.13382"}},"official":{"repos":["gyxxyg/vtg-llm"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/trafficvlm-a-controllable-visual-language","slug":"trafficvlm-a-controllable-visual-language","title":"TrafficVLM: A Controllable Visual Language Model for Traffic Video Captioning","date":"2024-04-14","arxiv_id":"2404.09275","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trafficvlm-a-controllable-visual-language#ran","syntology_url":"https://syntology.ai/paper/2404.09275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09275"}},"official":{"repos":["quangminhdinh/trafficvlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/do-you-remember-dense-video-captioning-with","slug":"do-you-remember-dense-video-captioning-with","title":"Do You Remember? Dense Video Captioning with Cross-Modal Memory Retrieval","date":"2024-04-11","arxiv_id":"2404.07610","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/do-you-remember-dense-video-captioning-with#ran","syntology_url":"https://syntology.ai/paper/2404.07610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07610"}},"official":{"repos":["ailab-kyunghee/cm2_dvc"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/omnivid-a-generative-framework-for-universal","slug":"omnivid-a-generative-framework-for-universal","title":"OmniVid: A Generative Framework for Universal Video Understanding","date":"2024-03-26","arxiv_id":"2403.17935","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/omnivid-a-generative-framework-for-universal#ran","syntology_url":"https://syntology.ai/paper/2403.17935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17935"}},"official":{"repos":["wangjk666/omnivid"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vtimellm-empower-llm-to-grasp-video-moments","slug":"vtimellm-empower-llm-to-grasp-video-moments","title":"VTimeLLM: Empower LLM to Grasp Video Moments","date":"2023-11-30","arxiv_id":"2311.18445","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vtimellm-empower-llm-to-grasp-video-moments#ran","syntology_url":"https://syntology.ai/paper/2311.18445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.18445"}},"official":{"repos":["huangb23/vtimellm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vidchapters-7m-video-chapters-at-scale","slug":"vidchapters-7m-video-chapters-at-scale","title":"VidChapters-7M: Video Chapters at Scale","date":"2023-09-25","arxiv_id":"2309.13952","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vidchapters-7m-video-chapters-at-scale#ran","syntology_url":"https://syntology.ai/paper/2309.13952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.13952"}},"official":null}},{"url":"/paper/soccernet-caption-dense-video-captioning-for","slug":"soccernet-caption-dense-video-captioning-for","title":"SoccerNet-Caption: Dense Video Captioning for Soccer Broadcasts Commentaries","date":"2023-04-10","arxiv_id":"2304.04565","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/soccernet-caption-dense-video-captioning-for#ran","syntology_url":"https://syntology.ai/paper/2304.04565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.04565"}},"official":null}},{"url":"/paper/learning-grounded-vision-language","slug":"learning-grounded-vision-language","title":"Learning Grounded Vision-Language Representation for Versatile Understanding in Untrimmed Videos","date":"2023-03-11","arxiv_id":"2303.06378","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-grounded-vision-language#ran","syntology_url":"https://syntology.ai/paper/2303.06378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06378"}},"official":{"repos":["zjr2000/gvl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-event-detection-and-captioning-as","slug":"unifying-event-detection-and-captioning-as","title":"Unifying Event Detection and Captioning as Sequence Generation via Pre-Training","date":"2022-07-18","arxiv_id":"2207.08625","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":4,"n_ran_checked":5,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/unifying-event-detection-and-captioning-as#ran","syntology_url":"https://syntology.ai/paper/2207.08625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.08625"}},"official":{"repos":["qiqang/uedvc"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-dense-video-captioning-with","slug":"end-to-end-dense-video-captioning-with","title":"End-to-End Dense Video Captioning with Parallel Decoding","date":"2021-08-17","arxiv_id":"2108.07781","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-dense-video-captioning-with#ran","syntology_url":"https://syntology.ai/paper/2108.07781","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.07781"}},"official":{"repos":["ttengwang/pdvc"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-better-use-of-audio-visual-cues-dense-video","slug":"a-better-use-of-audio-visual-cues-dense-video","title":"A Better Use of Audio-Visual Cues: Dense Video Captioning with Bi-modal Transformer","date":"2020-05-17","arxiv_id":"2005.08271","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-better-use-of-audio-visual-cues-dense-video#ran","syntology_url":"https://syntology.ai/paper/2005.08271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08271"}},"official":{"repos":["v-iashin/BMT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"0788b23a3fcaf32354c49e36f02e815176f227670da62898d17d3fc4992b5c51","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}