{"url":"/sota/video-captioning-on-msvd-1","task":{"name":"Video Captioning","url":"/task/video-captioning","note":null},"dataset":{"name":"MSVD","url":"/dataset/msvd"},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"**Video Captioning** is a task of automatic captioning a video by understanding the action and event in the video which can help in the retrieval of the video efficiently through text.\r\n\r\n\r\n<span class=\"description-source\">Source: [NITS-VC System for VATEX Video Captioning Challenge 2020 ](https://arxiv.org/abs/2006.04058)</span>","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["CIDEr","BLEU-4","METEOR","ROUGE-L","GS"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"CIDEr":null,"BLEU-4":"higher","METEOR":null,"ROUGE-L":"higher","GS":null}},"counts":{"rows":16,"rows_with_code":11,"rows_with_paper_page":16,"rows_dated":16,"rows_using_additional_data":7},"rows":[{"rank_in_archive_order":1,"model":"MaMMUT","metrics":{"CIDEr":"195.6"},"uses_additional_data":false,"paper_date":"2023-03-29","paper":"/paper/mammut-a-simple-architecture-for-joint","paper_url":"https://arxiv.org/abs/2303.16839v3","paper_title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","code":"https://github.com/lucidrains/mammut-pytorch","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":0,"n_samples":3,"n_pointer_only_licence":0}},{"rank_in_archive_order":2,"model":"VLAB","metrics":{"BLEU-4":"79.3","CIDEr":"179.8","METEOR":"51.2","ROUGE-L":"87.9"},"uses_additional_data":true,"paper_date":"2023-05-22","paper":"/paper/vlab-enhancing-video-language-pre-training-by","paper_url":"https://arxiv.org/abs/2305.13167v1","paper_title":"VLAB: Enhancing Video Language Pre-training by Feature Adapting and Blending","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":3,"model":"VALOR","metrics":{"BLEU-4":"80.7","CIDEr":"178.5","METEOR":"51.0","ROUGE-L":"87.9"},"uses_additional_data":true,"paper_date":"2023-04-17","paper":"/paper/valor-vision-audio-language-omni-perception","paper_url":"https://arxiv.org/abs/2304.08345v2","paper_title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","code":"https://github.com/TXH-mercury/VALOR","n_code_links":1,"syntology":null},{"rank_in_archive_order":4,"model":"COSA","metrics":{"BLEU-4":"76.5","CIDEr":"178.5"},"uses_additional_data":true,"paper_date":"2023-06-15","paper":"/paper/cosa-concatenated-sample-pretrained-vision","paper_url":"https://arxiv.org/abs/2306.09085v1","paper_title":"COSA: Concatenated Sample Pretrained Vision-Language Foundation Model","code":"https://github.com/txh-mercury/cosa","n_code_links":1,"syntology":null},{"rank_in_archive_order":5,"model":"mPLUG-2","metrics":{"BLEU-4":"70.5","CIDEr":"165.8","METEOR":"48.4","ROUGE-L":"85.3"},"uses_additional_data":false,"paper_date":"2023-02-01","paper":"/paper/mplug-2-a-modularized-multi-modal-foundation","paper_url":"https://arxiv.org/abs/2302.00402v1","paper_title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","code":"https://github.com/modelscope/modelscope","n_code_links":4,"syntology":{"n_ran":9,"n_unverified":10,"n_samples":19,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"HowToCaption","metrics":{"BLEU-4":"70.4","CIDEr":"154.2","METEOR":"46.4","ROUGE-L":"83.2"},"uses_additional_data":false,"paper_date":"2023-10-07","paper":"/paper/howtocaption-prompting-llms-to-transform","paper_url":"https://arxiv.org/abs/2310.04900v2","paper_title":"HowToCaption: Prompting LLMs to Transform Video Annotations at Scale","code":"https://github.com/ninatu/howtocaption","n_code_links":1,"syntology":{"n_ran":4,"n_unverified":1,"n_samples":5,"n_pointer_only_licence":5}},{"rank_in_archive_order":7,"model":"HiTeA","metrics":{"BLEU-4":"71.0","CIDEr":"146.9","METEOR":"45.3","ROUGE-L":"81.4"},"uses_additional_data":true,"paper_date":"2022-12-30","paper":"/paper/hitea-hierarchical-temporal-aware-video","paper_url":"https://arxiv.org/abs/2212.14546v1","paper_title":"HiTeA: Hierarchical Temporal-Aware Video-Language Pre-training","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":8,"model":"Vid2Seq","metrics":{"CIDEr":"146.2","METEOR":"45.3"},"uses_additional_data":true,"paper_date":"2023-02-27","paper":"/paper/vid2seq-large-scale-pretraining-of-a-visual","paper_url":"https://arxiv.org/abs/2302.14115v2","paper_title":"Vid2Seq: Large-Scale Pretraining of a Visual Language Model for Dense Video Captioning","code":"https://github.com/google-research/scenic/tree/main/scenic/projects/vid2seq","n_code_links":3,"syntology":null},{"rank_in_archive_order":9,"model":"VIOLETv2","metrics":{"CIDEr":"139.2"},"uses_additional_data":false,"paper_date":"2022-09-04","paper":"/paper/an-empirical-study-of-end-to-end-video","paper_url":"https://arxiv.org/abs/2209.01540v5","paper_title":"An Empirical Study of End-to-End Video-Language Transformers with Masked Visual Modeling","code":"https://github.com/tsujuifu/pytorch_empirical-mvm","n_code_links":1,"syntology":null},{"rank_in_archive_order":10,"model":"RTQ","metrics":{"BLEU-4":"66.9","CIDEr":"123.4","ROUGE-L":"82.2"},"uses_additional_data":false,"paper_date":"2023-12-01","paper":"/paper/rtq-rethinking-video-language-understanding","paper_url":"https://arxiv.org/abs/2312.00347v2","paper_title":"RTQ: Rethinking Video-language Understanding Based on Image-text Model","code":"https://github.com/SCZwangxiao/RTQ-MM2023","n_code_links":2,"syntology":null},{"rank_in_archive_order":11,"model":"CoCap (ViT/L14)","metrics":{"BLEU-4":"60.1","CIDEr":"121.5","METEOR":"41.4","ROUGE-L":"78.2"},"uses_additional_data":false,"paper_date":"2023-09-22","paper":"/paper/accurate-and-fast-compressed-video-captioning","paper_url":"https://arxiv.org/abs/2309.12867v2","paper_title":"Accurate and Fast Compressed Video Captioning","code":"https://github.com/acherstyx/CoCap","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":4,"n_samples":14,"n_pointer_only_licence":0}},{"rank_in_archive_order":12,"model":"VASTA (Vatex-backbone)","metrics":{"BLEU-4":"59.2","CIDEr":"119.7","METEOR":"40.65","ROUGE-L":"76.7"},"uses_additional_data":false,"paper_date":"2022-08-19","paper":"/paper/diverse-video-captioning-by-adaptive-spatio","paper_url":"https://arxiv.org/abs/2208.09266v1","paper_title":"Diverse Video Captioning by Adaptive Spatio-temporal Attention","code":"https://github.com/zohrehghaderi/vasta","n_code_links":1,"syntology":null},{"rank_in_archive_order":13,"model":"IcoCap (ViT-B/16)","metrics":{"BLEU-4":"59.1","CIDEr":"110.3","METEOR":"39.5","ROUGE-L":"76.5"},"uses_additional_data":true,"paper_date":"2023-10-05","paper":"/paper/icocap-improving-video-captioning-by","paper_url":"https://ieeexplore.ieee.org/abstract/document/10272675","paper_title":"IcoCap: Improving Video Captioning by Compounding Images","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":14,"model":"SEM-POS","metrics":{"BLEU-4":"60.1","CIDEr":"108.3","GS":"607.1","METEOR":"38.5","ROUGE-L":"76.0"},"uses_additional_data":false,"paper_date":"2023-03-26","paper":"/paper/sem-pos-grammatically-and-semantically","paper_url":"https://arxiv.org/abs/2303.14829v2","paper_title":"SEM-POS: Grammatically and Semantically Correct Video Captioning","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":15,"model":"VASTA (Kinetics-backbone)","metrics":{"BLEU-4":"56.1","CIDEr":"106.4","METEOR":"39.1","ROUGE-L":"74.5"},"uses_additional_data":false,"paper_date":"2022-08-19","paper":"/paper/diverse-video-captioning-by-adaptive-spatio","paper_url":"https://arxiv.org/abs/2208.09266v1","paper_title":"Diverse Video Captioning by Adaptive Spatio-temporal Attention","code":"https://github.com/zohrehghaderi/vasta","n_code_links":1,"syntology":null},{"rank_in_archive_order":16,"model":"IcoCap (ViT-B/32)","metrics":{"BLEU-4":"56.3","CIDEr":"103.8","METEOR":"38.9","ROUGE-L":"75.0"},"uses_additional_data":true,"paper_date":"2023-10-05","paper":"/paper/icocap-improving-video-captioning-by","paper_url":"https://ieeexplore.ieee.org/abstract/document/10272675","paper_title":"IcoCap: Improving Video Captioning by Compounding Images","code":null,"n_code_links":0,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":4,"rows_with_any_sample_ran":4,"distinct_papers_with_graph_line":4,"distinct_papers_with_any_sample_ran":4,"samples_over_distinct_papers":{"n_ran":26,"n_unverified":15,"n_samples":41,"n_pointer_only_licence":5,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":26,"n_unverified":15,"n_samples":41,"n_pointer_only_licence":5,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}