{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-captioning/papers/ran/1","list_of":"/task/video-captioning","task":"Video Captioning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,64],"of":64,"counts":{"archive_papers_tagged":473,"with_a_code_link":211,"where_syntology_ran_a_sample":64,"not_listed_spam_title":0,"listed":473,"listed_where_code_ran":64,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":56,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":56,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-captioning/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/video-salmonn-2-captioning-enhanced-audio","slug":"video-salmonn-2-captioning-enhanced-audio","title":"video-SALMONN 2: Captioning-Enhanced Audio-Visual Large Language Models","date":"2025-06-18","arxiv_id":"2506.15220","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-salmonn-2-captioning-enhanced-audio#ran","syntology_url":"https://syntology.ai/paper/2506.15220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.15220"}},"official":{"repos":["bytedance/video-salmonn-2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/caption-anything-in-video-fine-grained-object","slug":"caption-anything-in-video-fine-grained-object","title":"Caption Anything in Video: Fine-grained Object-centric Captioning via Spatiotemporal Multimodal Prompting","date":"2025-04-07","arxiv_id":"2504.05541","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/caption-anything-in-video-fine-grained-object#ran","syntology_url":"https://syntology.ai/paper/2504.05541","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05541"}},"official":{"repos":["yunlong10/CAT-V"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/pretrained-image-text-models-are-secretly","slug":"pretrained-image-text-models-are-secretly","title":"Pretrained Image-Text Models are Secretly Video Captioners","date":"2025-02-19","arxiv_id":"2502.13363","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pretrained-image-text-models-are-secretly#ran","syntology_url":"https://syntology.ai/paper/2502.13363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13363"}},"official":{"repos":["chunhuizng/mllm-video-captioner"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/g-veval-a-versatile-metric-for-evaluating","slug":"g-veval-a-versatile-metric-for-evaluating","title":"G-VEval: A Versatile Metric for Evaluating Image and Video Captions Using GPT-4o","date":"2024-12-18","arxiv_id":"2412.13647","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/g-veval-a-versatile-metric-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2412.13647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13647"}},"official":{"repos":["ztangaj/gveval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/implicit-location-caption-alignment-via","slug":"implicit-location-caption-alignment-via","title":"Implicit Location-Caption Alignment via Complementary Masking for Weakly-Supervised Dense Video Captioning","date":"2024-12-17","arxiv_id":"2412.12791","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/implicit-location-caption-alignment-via#ran","syntology_url":"https://syntology.ai/paper/2412.12791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12791"}},"official":{"repos":["ShipingGe/ILCACM"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/videollm-knows-when-to-speak-enhancing-time","slug":"videollm-knows-when-to-speak-enhancing-time","title":"VideoLLM Knows When to Speak: Enhancing Time-Sensitive Video Comprehension with Video-Text Duet Interaction Format","date":"2024-11-27","arxiv_id":"2411.17991","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videollm-knows-when-to-speak-enhancing-time#ran","syntology_url":"https://syntology.ai/paper/2411.17991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17991"}},"official":{"repos":["yellow-binary-tree/mmduet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lvd-2m-a-long-take-video-dataset-with","slug":"lvd-2m-a-long-take-video-dataset-with","title":"LVD-2M: A Long-take Video Dataset with Temporally Dense Captions","date":"2024-10-14","arxiv_id":"2410.10816","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lvd-2m-a-long-take-video-dataset-with#ran","syntology_url":"https://syntology.ai/paper/2410.10816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10816"}},"official":{"repos":["silentview/lvd-2m"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grounded-videollm-sharpening-fine-grained","slug":"grounded-videollm-sharpening-fine-grained","title":"Grounded-VideoLLM: Sharpening Fine-grained Temporal Grounding in Video Large Language Models","date":"2024-10-04","arxiv_id":"2410.03290","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounded-videollm-sharpening-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2410.03290","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03290"}},"official":{"repos":["whb139426/grounded-video-llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cogvideox-text-to-video-diffusion-models-with","slug":"cogvideox-text-to-video-diffusion-models-with","title":"CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer","date":"2024-08-12","arxiv_id":"2408.06072","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cogvideox-text-to-video-diffusion-models-with#ran","syntology_url":"https://syntology.ai/paper/2408.06072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.06072"}},"official":{"repos":["thudm/cogvideo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2407-21757","slug":"2407-21757","title":"Learning Video Context as Interleaved Multimodal Sequences","date":"2024-07-31","arxiv_id":"2407.21757","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2407-21757#ran","syntology_url":"https://syntology.ai/paper/2407.21757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21757"}},"official":{"repos":["showlab/movieseq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tarsier-recipes-for-training-and-evaluating-1","slug":"tarsier-recipes-for-training-and-evaluating-1","title":"Tarsier: Recipes for Training and Evaluating Large Video Description Models","date":"2024-06-30","arxiv_id":"2407.00634","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier-recipes-for-training-and-evaluating-1#ran","syntology_url":"https://syntology.ai/paper/2407.00634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00634"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/alanavlm-a-multimodal-embodied-ai-foundation","slug":"alanavlm-a-multimodal-embodied-ai-foundation","title":"AlanaVLM: A Multimodal Embodied AI Foundation Model for Egocentric Video Understanding","date":"2024-06-19","arxiv_id":"2406.13807","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alanavlm-a-multimodal-embodied-ai-foundation#ran","syntology_url":"https://syntology.ai/paper/2406.13807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13807"}},"official":{"repos":["alanaai/evud"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videogpt-integrating-image-and-video-encoders","slug":"videogpt-integrating-image-and-video-encoders","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","date":"2024-06-13","arxiv_id":"2406.09418","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videogpt-integrating-image-and-video-encoders#ran","syntology_url":"https://syntology.ai/paper/2406.09418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09418"}},"official":{"repos":["mbzuai-oryx/videogpt-plus"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/videollama-2-advancing-spatial-temporal","slug":"videollama-2-advancing-spatial-temporal","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","date":"2024-06-11","arxiv_id":"2406.07476","repositories_listed":3,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":7,"n_instrument":5,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":8,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/videollama-2-advancing-spatial-temporal#ran","syntology_url":"https://syntology.ai/paper/2406.07476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07476"}},"official":{"repos":["damo-nlp-sg/videollama2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/vript-a-video-is-worth-thousands-of-words","slug":"vript-a-video-is-worth-thousands-of-words","title":"Vript: A Video Is Worth Thousands of Words","date":"2024-06-10","arxiv_id":"2406.06040","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vript-a-video-is-worth-thousands-of-words#ran","syntology_url":"https://syntology.ai/paper/2406.06040","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06040"}},"official":{"repos":["mutonix/vript"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vtg-llm-integrating-timestamp-knowledge-into","slug":"vtg-llm-integrating-timestamp-knowledge-into","title":"VTG-LLM: Integrating Timestamp Knowledge into Video LLMs for Enhanced Video Temporal Grounding","date":"2024-05-22","arxiv_id":"2405.13382","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vtg-llm-integrating-timestamp-knowledge-into#ran","syntology_url":"https://syntology.ai/paper/2405.13382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.13382"}},"official":{"repos":["gyxxyg/vtg-llm"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/trafficvlm-a-controllable-visual-language","slug":"trafficvlm-a-controllable-visual-language","title":"TrafficVLM: A Controllable Visual Language Model for Traffic Video Captioning","date":"2024-04-14","arxiv_id":"2404.09275","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trafficvlm-a-controllable-visual-language#ran","syntology_url":"https://syntology.ai/paper/2404.09275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09275"}},"official":{"repos":["quangminhdinh/trafficvlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/do-you-remember-dense-video-captioning-with","slug":"do-you-remember-dense-video-captioning-with","title":"Do You Remember? Dense Video Captioning with Cross-Modal Memory Retrieval","date":"2024-04-11","arxiv_id":"2404.07610","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/do-you-remember-dense-video-captioning-with#ran","syntology_url":"https://syntology.ai/paper/2404.07610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07610"}},"official":{"repos":["ailab-kyunghee/cm2_dvc"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ma-lmm-memory-augmented-large-multimodal","slug":"ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","arxiv_id":"2404.05726","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ma-lmm-memory-augmented-large-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.05726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05726"}},"official":{"repos":["boheumd/MA-LMM"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/omnivid-a-generative-framework-for-universal","slug":"omnivid-a-generative-framework-for-universal","title":"OmniVid: A Generative Framework for Universal Video Understanding","date":"2024-03-26","arxiv_id":"2403.17935","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/omnivid-a-generative-framework-for-universal#ran","syntology_url":"https://syntology.ai/paper/2403.17935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17935"}},"official":{"repos":["wangjk666/omnivid"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/video-recap-recursive-captioning-of-hour-long","slug":"video-recap-recursive-captioning-of-hour-long","title":"Video ReCap: Recursive Captioning of Hour-Long Videos","date":"2024-02-20","arxiv_id":"2402.13250","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-recap-recursive-captioning-of-hour-long#ran","syntology_url":"https://syntology.ai/paper/2402.13250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13250"}},"official":{"repos":["md-mohaiminul/VideoRecap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/connect-collapse-corrupt-learning-cross-modal","slug":"connect-collapse-corrupt-learning-cross-modal","title":"Connect, Collapse, Corrupt: Learning Cross-Modal Tasks with Uni-Modal Data","date":"2024-01-16","arxiv_id":"2401.08567","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/connect-collapse-corrupt-learning-cross-modal#ran","syntology_url":"https://syntology.ai/paper/2401.08567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08567"}},"official":{"repos":["yuhui-zh15/c3"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/shot2story20k-a-new-benchmark-for","slug":"shot2story20k-a-new-benchmark-for","title":"Shot2Story20K: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","date":"2023-12-16","arxiv_id":"2312.10300","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/shot2story20k-a-new-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2312.10300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10300"}},"official":{"repos":["bytedance/Shot2Story"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vtimellm-empower-llm-to-grasp-video-moments","slug":"vtimellm-empower-llm-to-grasp-video-moments","title":"VTimeLLM: Empower LLM to Grasp Video Moments","date":"2023-11-30","arxiv_id":"2311.18445","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vtimellm-empower-llm-to-grasp-video-moments#ran","syntology_url":"https://syntology.ai/paper/2311.18445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.18445"}},"official":{"repos":["huangb23/vtimellm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/howtocaption-prompting-llms-to-transform","slug":"howtocaption-prompting-llms-to-transform","title":"HowToCaption: Prompting LLMs to Transform Video Annotations at Scale","date":"2023-10-07","arxiv_id":"2310.04900","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/howtocaption-prompting-llms-to-transform#ran","syntology_url":"https://syntology.ai/paper/2310.04900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04900"}},"official":{"repos":["ninatu/howtocaption"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vidchapters-7m-video-chapters-at-scale","slug":"vidchapters-7m-video-chapters-at-scale","title":"VidChapters-7M: Video Chapters at Scale","date":"2023-09-25","arxiv_id":"2309.13952","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vidchapters-7m-video-chapters-at-scale#ran","syntology_url":"https://syntology.ai/paper/2309.13952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.13952"}},"official":null}},{"url":"/paper/accurate-and-fast-compressed-video-captioning","slug":"accurate-and-fast-compressed-video-captioning","title":"Accurate and Fast Compressed Video Captioning","date":"2023-09-22","arxiv_id":"2309.12867","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":8,"n_ran_checked":9,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/accurate-and-fast-compressed-video-captioning#ran","syntology_url":"https://syntology.ai/paper/2309.12867","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12867"}},"official":{"repos":["acherstyx/CoCap"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":8,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multicapclip-auto-encoding-prompts-for-zero","slug":"multicapclip-auto-encoding-prompts-for-zero","title":"MultiCapCLIP: Auto-Encoding Prompts for Zero-Shot Multilingual Visual Captioning","date":"2023-08-25","arxiv_id":"2308.13218","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/multicapclip-auto-encoding-prompts-for-zero#ran","syntology_url":"https://syntology.ai/paper/2308.13218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.13218"}},"official":{"repos":["yangbang18/multicapclip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-pet-vision-and-language-parameter","slug":"vl-pet-vision-and-language-parameter","title":"VL-PET: Vision-and-Language Parameter-Efficient Tuning via Granularity Control","date":"2023-08-18","arxiv_id":"2308.09804","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vl-pet-vision-and-language-parameter#ran","syntology_url":"https://syntology.ai/paper/2308.09804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09804"}},"official":{"repos":["henryhzy/vl-pet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompt-switch-efficient-clip-adaptation-for","slug":"prompt-switch-efficient-clip-adaptation-for","title":"Prompt Switch: Efficient CLIP Adaptation for Text-Video Retrieval","date":"2023-08-15","arxiv_id":"2308.07648","repositories_listed":1,"syntology":{"n":16,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":12,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":16,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 12 unverified","sample_list":"/paper/prompt-switch-efficient-clip-adaptation-for#ran","syntology_url":"https://syntology.ai/paper/2308.07648","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07648"}},"official":{"repos":["bladewaltz1/promptswitch"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":12,"ran_from_kinds":["official"]}}},{"url":"/paper/youku-mplug-a-10-million-large-scale-chinese","slug":"youku-mplug-a-10-million-large-scale-chinese","title":"Youku-mPLUG: A 10 Million Large-scale Chinese Video-Language Dataset for Pre-training and Benchmarks","date":"2023-06-07","arxiv_id":"2306.04362","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/youku-mplug-a-10-million-large-scale-chinese#ran","syntology_url":"https://syntology.ai/paper/2306.04362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.04362"}},"official":{"repos":["x-plug/youku-mplug"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vast-a-vision-audio-subtitle-text-omni-1","slug":"vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","arxiv_id":"2305.18500","repositories_listed":2,"syntology":{"n":42,"n_ran":35,"n_constructed":4,"n_ran_checked":29,"n_instrument":6,"n_unverified":7,"n_honours":2,"n_violates":1,"n_no_contract":26,"n_pointer_only":8,"phrase":"35 ran (of which 4 constructed an object rather than computing a result; 29 with no instrument failure: 2 honoured, 1 violated, 26 with no contract checked; 6 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/vast-a-vision-audio-subtitle-text-omni-1#ran","syntology_url":"https://syntology.ai/paper/2305.18500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18500"}},"official":{"repos":["txh-mercury/vast"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":4,"n_ran_no_instrument_failure":12,"n_unverified":7,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/pali-x-on-scaling-up-a-multilingual-vision","slug":"pali-x-on-scaling-up-a-multilingual-vision","title":"PaLI-X: On Scaling up a Multilingual Vision and Language Model","date":"2023-05-29","arxiv_id":"2305.18565","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":4,"n_no_contract":1,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 4 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pali-x-on-scaling-up-a-multilingual-vision#ran","syntology_url":"https://syntology.ai/paper/2305.18565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18565"}},"official":null}},{"url":"/paper/soccernet-caption-dense-video-captioning-for","slug":"soccernet-caption-dense-video-captioning-for","title":"SoccerNet-Caption: Dense Video Captioning for Soccer Broadcasts Commentaries","date":"2023-04-10","arxiv_id":"2304.04565","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/soccernet-caption-dense-video-captioning-for#ran","syntology_url":"https://syntology.ai/paper/2304.04565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.04565"}},"official":null}},{"url":"/paper/hierarchical-video-moment-retrieval-and-step","slug":"hierarchical-video-moment-retrieval-and-step","title":"Hierarchical Video-Moment Retrieval and Step-Captioning","date":"2023-03-29","arxiv_id":"2303.16406","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hierarchical-video-moment-retrieval-and-step#ran","syntology_url":"https://syntology.ai/paper/2303.16406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16406"}},"official":{"repos":["j-min/HiREST"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mammut-a-simple-architecture-for-joint","slug":"mammut-a-simple-architecture-for-joint","title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","date":"2023-03-29","arxiv_id":"2303.16839","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mammut-a-simple-architecture-for-joint#ran","syntology_url":"https://syntology.ai/paper/2303.16839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16839"}},"official":null}},{"url":"/paper/fine-grained-audible-video-description","slug":"fine-grained-audible-video-description","title":"Fine-grained Audible Video Description","date":"2023-03-27","arxiv_id":"2303.15616","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/fine-grained-audible-video-description#ran","syntology_url":"https://syntology.ai/paper/2303.15616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.15616"}},"official":{"repos":["opennlplab/favdbench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/meltr-meta-loss-transformer-for-learning-to","slug":"meltr-meta-loss-transformer-for-learning-to","title":"MELTR: Meta Loss Transformer for Learning to Fine-tune Video Foundation Models","date":"2023-03-23","arxiv_id":"2303.13009","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/meltr-meta-loss-transformer-for-learning-to#ran","syntology_url":"https://syntology.ai/paper/2303.13009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13009"}},"official":{"repos":["mlvlab/MELTR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/positive-augmented-constrastive-learning-for","slug":"positive-augmented-constrastive-learning-for","title":"Positive-Augmented Contrastive Learning for Image and Video Captioning Evaluation","date":"2023-03-21","arxiv_id":"2303.12112","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/positive-augmented-constrastive-learning-for#ran","syntology_url":"https://syntology.ai/paper/2303.12112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.12112"}},"official":{"repos":["aimagelab/pacscore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-grounded-vision-language","slug":"learning-grounded-vision-language","title":"Learning Grounded Vision-Language Representation for Versatile Understanding in Untrimmed Videos","date":"2023-03-11","arxiv_id":"2303.06378","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-grounded-vision-language#ran","syntology_url":"https://syntology.ai/paper/2303.06378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06378"}},"official":{"repos":["zjr2000/gvl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mplug-2-a-modularized-multi-modal-foundation","slug":"mplug-2-a-modularized-multi-modal-foundation","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","date":"2023-02-01","arxiv_id":"2302.00402","repositories_listed":4,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":9,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mplug-2-a-modularized-multi-modal-foundation#ran","syntology_url":"https://syntology.ai/paper/2302.00402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00402"}},"official":{"repos":["alibaba/AliceMind"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/cap4video-what-can-auxiliary-captions-do-for","slug":"cap4video-what-can-auxiliary-captions-do-for","title":"Cap4Video: What Can Auxiliary Captions Do for Text-Video Retrieval?","date":"2022-12-31","arxiv_id":"2301.00184","repositories_listed":4,"syntology":{"n":25,"n_ran":20,"n_constructed":0,"n_ran_checked":12,"n_instrument":8,"n_unverified":5,"n_honours":2,"n_violates":1,"n_no_contract":9,"n_pointer_only":11,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 2 honoured, 1 violated, 9 with no contract checked; 8 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/cap4video-what-can-auxiliary-captions-do-for#ran","syntology_url":"https://syntology.ai/paper/2301.00184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.00184"}},"official":{"repos":["whwu95/Cap4Video"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/expectation-maximization-contrastive-learning","slug":"expectation-maximization-contrastive-learning","title":"Expectation-Maximization Contrastive Learning for Compact Video-and-Language Representations","date":"2022-11-21","arxiv_id":"2211.11427","repositories_listed":4,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/expectation-maximization-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2211.11427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11427"}},"official":{"repos":["jpthu17/emcl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/why-is-winoground-hard-investigating-failures","slug":"why-is-winoground-hard-investigating-failures","title":"Why is Winoground Hard? Investigating Failures in Visuolinguistic Compositionality","date":"2022-11-01","arxiv_id":"2211.00768","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/why-is-winoground-hard-investigating-failures#ran","syntology_url":"https://syntology.ai/paper/2211.00768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00768"}},"official":{"repos":["ajd12342/why-winoground-hard"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/storydall-e-adapting-pretrained-text-to-image","slug":"storydall-e-adapting-pretrained-text-to-image","title":"StoryDALL-E: Adapting Pretrained Text-to-Image Transformers for Story Continuation","date":"2022-09-13","arxiv_id":"2209.06192","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/storydall-e-adapting-pretrained-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2209.06192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.06192"}},"official":{"repos":["adymaharana/storydalle"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-event-detection-and-captioning-as","slug":"unifying-event-detection-and-captioning-as","title":"Unifying Event Detection and Captioning as Sequence Generation via Pre-Training","date":"2022-07-18","arxiv_id":"2207.08625","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":4,"n_ran_checked":5,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/unifying-event-detection-and-captioning-as#ran","syntology_url":"https://syntology.ai/paper/2207.08625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.08625"}},"official":{"repos":["qiqang/uedvc"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/git-a-generative-image-to-text-transformer","slug":"git-a-generative-image-to-text-transformer","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","date":"2022-05-27","arxiv_id":"2205.14100","repositories_listed":1,"syntology":{"n":21,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/git-a-generative-image-to-text-transformer#ran","syntology_url":"https://syntology.ai/paper/2205.14100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14100"}},"official":{"repos":["microsoft/GenerativeImage2Text"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/gl-rg-global-local-representation-granularity","slug":"gl-rg-global-local-representation-granularity","title":"GL-RG: Global-Local Representation Granularity for Video Captioning","date":"2022-05-22","arxiv_id":"2205.10706","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/gl-rg-global-local-representation-granularity#ran","syntology_url":"https://syntology.ai/paper/2205.10706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10706"}},"official":{"repos":["ylqi/gl-rg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-with-image-descriptors-are","slug":"language-models-with-image-descriptors-are","title":"Language Models with Image Descriptors are Strong Few-Shot Video-Language Learners","date":"2022-05-22","arxiv_id":"2205.10747","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-with-image-descriptors-are#ran","syntology_url":"https://syntology.ai/paper/2205.10747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10747"}},"official":{"repos":["mikewangwzhl/vidil"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/swinbert-end-to-end-transformers-with-sparse","slug":"swinbert-end-to-end-transformers-with-sparse","title":"SwinBERT: End-to-End Transformers with Sparse Attention for Video Captioning","date":"2021-11-25","arxiv_id":"2111.13196","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swinbert-end-to-end-transformers-with-sparse#ran","syntology_url":"https://syntology.ai/paper/2111.13196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.13196"}},"official":{"repos":["microsoft/swinbert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/emscore-evaluating-video-captioning-via","slug":"emscore-evaluating-video-captioning-via","title":"EMScore: Evaluating Video Captioning via Coarse-Grained and Fine-Grained Embedding Matching","date":"2021-11-17","arxiv_id":"2111.08919","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emscore-evaluating-video-captioning-via#ran","syntology_url":"https://syntology.ai/paper/2111.08919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08919"}},"official":{"repos":["shiyaya/emscore"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-dense-video-captioning-with","slug":"end-to-end-dense-video-captioning-with","title":"End-to-End Dense Video Captioning with Parallel Decoding","date":"2021-08-17","arxiv_id":"2108.07781","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-dense-video-captioning-with#ran","syntology_url":"https://syntology.ai/paper/2108.07781","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.07781"}},"official":{"repos":["ttengwang/pdvc"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/value-a-multi-task-benchmark-for-video-and","slug":"value-a-multi-task-benchmark-for-video-and","title":"VALUE: A Multi-Task Benchmark for Video-and-Language Understanding Evaluation","date":"2021-06-08","arxiv_id":"2106.04632","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/value-a-multi-task-benchmark-for-video-and#ran","syntology_url":"https://syntology.ai/paper/2106.04632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04632"}},"official":{"repos":["VALUE-Leaderboard/StarterCode"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-generation-and-evaluation-of-visual","slug":"improving-generation-and-evaluation-of-visual","title":"Improving Generation and Evaluation of Visual Stories via Semantic Consistency","date":"2021-05-20","arxiv_id":"2105.10026","repositories_listed":1,"syntology":{"n":14,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/improving-generation-and-evaluation-of-visual#ran","syntology_url":"https://syntology.ai/paper/2105.10026","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.10026"}},"official":{"repos":["adymaharana/StoryViz"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/frozen-in-time-a-joint-video-and-image","slug":"frozen-in-time-a-joint-video-and-image","title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval","date":"2021-04-01","arxiv_id":"2104.00650","repositories_listed":5,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/frozen-in-time-a-joint-video-and-image#ran","syntology_url":"https://syntology.ai/paper/2104.00650","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.00650"}},"official":{"repos":["m-bain/frozen-in-time"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/coot-cooperative-hierarchical-transformer-for","slug":"coot-cooperative-hierarchical-transformer-for","title":"COOT: Cooperative Hierarchical Transformer for Video-Text Representation Learning","date":"2020-11-01","arxiv_id":"2011.00597","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coot-cooperative-hierarchical-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2011.00597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00597"}},"official":{"repos":["gingsi/coot-videotext"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-better-use-of-audio-visual-cues-dense-video","slug":"a-better-use-of-audio-visual-cues-dense-video","title":"A Better Use of Audio-Visual Cues: Dense Video Captioning with Bi-modal Transformer","date":"2020-05-17","arxiv_id":"2005.08271","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-better-use-of-audio-visual-cues-dense-video#ran","syntology_url":"https://syntology.ai/paper/2005.08271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08271"}},"official":{"repos":["v-iashin/BMT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mart-memory-augmented-recurrent-transformer","slug":"mart-memory-augmented-recurrent-transformer","title":"MART: Memory-Augmented Recurrent Transformer for Coherent Video Paragraph Captioning","date":"2020-05-11","arxiv_id":"2005.05402","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/mart-memory-augmented-recurrent-transformer#ran","syntology_url":"https://syntology.ai/paper/2005.05402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05402"}},"official":{"repos":["jayleicn/recurrent-transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hero-hierarchical-encoder-for-video-language","slug":"hero-hierarchical-encoder-for-video-language","title":"HERO: Hierarchical Encoder for Video+Language Omni-representation Pre-training","date":"2020-05-01","arxiv_id":"2005.00200","repositories_listed":3,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hero-hierarchical-encoder-for-video-language#ran","syntology_url":"https://syntology.ai/paper/2005.00200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.00200"}},"official":{"repos":["linjieli222/HERO"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/video2commonsense-generating-commonsense","slug":"video2commonsense-generating-commonsense","title":"Video2Commonsense: Generating Commonsense Descriptions to Enrich Video Captioning","date":"2020-03-11","arxiv_id":"2003.05162","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video2commonsense-generating-commonsense#ran","syntology_url":"https://syntology.ai/paper/2003.05162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05162"}},"official":{"repos":["jacobswan1/Video2Commonsense"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/univilm-a-unified-video-and-language-pre","slug":"univilm-a-unified-video-and-language-pre","title":"UniVL: A Unified Video and Language Pre-Training Model for Multimodal Understanding and Generation","date":"2020-02-15","arxiv_id":"2002.06353","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/univilm-a-unified-video-and-language-pre#ran","syntology_url":"https://syntology.ai/paper/2002.06353","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06353"}},"official":{"repos":["microsoft/UniVL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/continual-and-multi-task-architecture-search","slug":"continual-and-multi-task-architecture-search","title":"Continual and Multi-Task Architecture Search","date":"2019-06-12","arxiv_id":"1906.05226","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/continual-and-multi-task-architecture-search#ran","syntology_url":"https://syntology.ai/paper/1906.05226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.05226"}},"official":{"repos":["ramakanth-pasunuru/CAS-MAS"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/videobert-a-joint-model-for-video-and","slug":"videobert-a-joint-model-for-video-and","title":"VideoBERT: A Joint Model for Video and Language Representation Learning","date":"2019-04-03","arxiv_id":"1904.01766","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videobert-a-joint-model-for-video-and#ran","syntology_url":"https://syntology.ai/paper/1904.01766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.01766"}},"official":null}},{"url":"/paper/reconstruction-network-for-video-captioning","slug":"reconstruction-network-for-video-captioning","title":"Reconstruction Network for Video Captioning","date":"2018-03-30","arxiv_id":"1803.11438","repositories_listed":3,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reconstruction-network-for-video-captioning#ran","syntology_url":"https://syntology.ai/paper/1803.11438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.11438"}},"official":null}}],"record_sha256":"76437511e39fb25d7e88220568e66080f72149e011cddc6e3453880ea6e4fb61","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}