{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-captioning/papers/2","list_of":"/task/video-captioning","task":"Video Captioning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":5,"rows_per_page":100,"rows":[101,200],"of":473,"counts":{"archive_papers_tagged":473,"with_a_code_link":211,"where_syntology_ran_a_sample":64,"not_listed_spam_title":0,"listed":473,"listed_where_code_ran":64,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":56,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":56,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-captioning","prev":"/task/video-captioning","next":"/task/video-captioning/papers/3","papers":[{"url":"/paper/edit-as-you-wish-video-description-editing","slug":"edit-as-you-wish-video-description-editing","title":"Edit As You Wish: Video Caption Editing with Multi-grained User Control","date":"2023-05-15","arxiv_id":"2305.08389","repositories_listed":1,"syntology":null},{"url":"/paper/visual-transformation-telling","slug":"visual-transformation-telling","title":"Visual Transformation Telling","date":"2023-05-03","arxiv_id":"2305.01928","repositories_listed":1,"syntology":null},{"url":"/paper/from-association-to-generation-text-only","slug":"from-association-to-generation-text-only","title":"From Association to Generation: Text-only Captioning by Unsupervised Cross-modal Mapping","date":"2023-04-26","arxiv_id":"2304.13273","repositories_listed":1,"syntology":null},{"url":"/paper/valor-vision-audio-language-omni-perception","slug":"valor-vision-audio-language-omni-perception","title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","date":"2023-04-17","arxiv_id":"2304.08345","repositories_listed":1,"syntology":null},{"url":"/paper/video-chatcaptioner-towards-the-enriched","slug":"video-chatcaptioner-towards-the-enriched","title":"Video ChatCaptioner: Towards Enriched Spatiotemporal Descriptions","date":"2023-04-09","arxiv_id":"2304.04227","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-video-moment-retrieval-and-step","slug":"hierarchical-video-moment-retrieval-and-step","title":"Hierarchical Video-Moment Retrieval and Step-Captioning","date":"2023-03-29","arxiv_id":"2303.16406","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hierarchical-video-moment-retrieval-and-step#ran","syntology_url":"https://syntology.ai/paper/2303.16406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16406"}},"official":{"repos":["j-min/HiREST"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mammut-a-simple-architecture-for-joint","slug":"mammut-a-simple-architecture-for-joint","title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","date":"2023-03-29","arxiv_id":"2303.16839","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mammut-a-simple-architecture-for-joint#ran","syntology_url":"https://syntology.ai/paper/2303.16839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16839"}},"official":null}},{"url":"/paper/fine-grained-audible-video-description","slug":"fine-grained-audible-video-description","title":"Fine-grained Audible Video Description","date":"2023-03-27","arxiv_id":"2303.15616","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/fine-grained-audible-video-description#ran","syntology_url":"https://syntology.ai/paper/2303.15616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.15616"}},"official":{"repos":["opennlplab/favdbench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/goal-a-challenging-knowledge-grounded-video","slug":"goal-a-challenging-knowledge-grounded-video","title":"GOAL: A Challenging Knowledge-grounded Video Captioning Benchmark for Real-time Soccer Commentary Generation","date":"2023-03-26","arxiv_id":"2303.14655","repositories_listed":1,"syntology":null},{"url":"/paper/meltr-meta-loss-transformer-for-learning-to","slug":"meltr-meta-loss-transformer-for-learning-to","title":"MELTR: Meta Loss Transformer for Learning to Fine-tune Video Foundation Models","date":"2023-03-23","arxiv_id":"2303.13009","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/meltr-meta-loss-transformer-for-learning-to#ran","syntology_url":"https://syntology.ai/paper/2303.13009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13009"}},"official":{"repos":["mlvlab/MELTR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/positive-augmented-constrastive-learning-for","slug":"positive-augmented-constrastive-learning-for","title":"Positive-Augmented Contrastive Learning for Image and Video Captioning Evaluation","date":"2023-03-21","arxiv_id":"2303.12112","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/positive-augmented-constrastive-learning-for#ran","syntology_url":"https://syntology.ai/paper/2303.12112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.12112"}},"official":{"repos":["aimagelab/pacscore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/action-knowledge-for-video-captioning-with","slug":"action-knowledge-for-video-captioning-with","title":"Action knowledge for video captioning with graph neural networks","date":"2023-03-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/accommodating-audio-modality-in-clip-for","slug":"accommodating-audio-modality-in-clip-for","title":"Accommodating Audio Modality in CLIP for Multimodal Processing","date":"2023-03-12","arxiv_id":"2303.06591","repositories_listed":1,"syntology":null},{"url":"/paper/learning-grounded-vision-language","slug":"learning-grounded-vision-language","title":"Learning Grounded Vision-Language Representation for Versatile Understanding in Untrimmed Videos","date":"2023-03-11","arxiv_id":"2303.06378","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-grounded-vision-language#ran","syntology_url":"https://syntology.ai/paper/2303.06378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06378"}},"official":{"repos":["zjr2000/gvl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/meteor-guided-divergence-for-video-captioning","slug":"meteor-guided-divergence-for-video-captioning","title":"METEOR Guided Divergence for Video Captioning","date":"2022-12-20","arxiv_id":"2212.10690","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-explainable-video-representation","slug":"contextual-explainable-video-representation","title":"Contextual Explainable Video Representation: Human Perception-based Understanding","date":"2022-12-12","arxiv_id":"2212.06206","repositories_listed":1,"syntology":null},{"url":"/paper/refined-semantic-enhancement-towards","slug":"refined-semantic-enhancement-towards","title":"Refined Semantic Enhancement towards Frequency Diffusion for Video Captioning","date":"2022-11-28","arxiv_id":"2211.15076","repositories_listed":1,"syntology":null},{"url":"/paper/vltint-visual-linguistic-transformer-in","slug":"vltint-visual-linguistic-transformer-in","title":"VLTinT: Visual-Linguistic Transformer-in-Transformer for Coherent Video Paragraph Captioning","date":"2022-11-28","arxiv_id":"2211.15103","repositories_listed":1,"syntology":null},{"url":"/paper/visual-commonsense-aware-representation","slug":"visual-commonsense-aware-representation","title":"Visual Commonsense-aware Representation Network for Video Captioning","date":"2022-11-17","arxiv_id":"2211.09469","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-metadata-extraction-from-dense-video","slug":"semantic-metadata-extraction-from-dense-video","title":"Event and Entity Extraction from Generated Video Captions","date":"2022-11-05","arxiv_id":"2211.02982","repositories_listed":1,"syntology":null},{"url":"/paper/why-is-winoground-hard-investigating-failures","slug":"why-is-winoground-hard-investigating-failures","title":"Why is Winoground Hard? Investigating Failures in Visuolinguistic Compositionality","date":"2022-11-01","arxiv_id":"2211.00768","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/why-is-winoground-hard-investigating-failures#ran","syntology_url":"https://syntology.ai/paper/2211.00768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00768"}},"official":{"repos":["ajd12342/why-winoground-hard"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-pre-training-basics-recent","slug":"vision-language-pre-training-basics-recent","title":"Vision-Language Pre-training: Basics, Recent Advances, and Future Trends","date":"2022-10-17","arxiv_id":"2210.09263","repositories_listed":1,"syntology":null},{"url":"/paper/thinking-hallucination-for-video-captioning","slug":"thinking-hallucination-for-video-captioning","title":"Thinking Hallucination for Video Captioning","date":"2022-09-28","arxiv_id":"2209.13853","repositories_listed":1,"syntology":null},{"url":"/paper/storydall-e-adapting-pretrained-text-to-image","slug":"storydall-e-adapting-pretrained-text-to-image","title":"StoryDALL-E: Adapting Pretrained Text-to-Image Transformers for Story Continuation","date":"2022-09-13","arxiv_id":"2209.06192","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/storydall-e-adapting-pretrained-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2209.06192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.06192"}},"official":{"repos":["adymaharana/storydalle"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-of-end-to-end-video","slug":"an-empirical-study-of-end-to-end-video","title":"An Empirical Study of End-to-End Video-Language Transformers with Masked Visual Modeling","date":"2022-09-04","arxiv_id":"2209.01540","repositories_listed":1,"syntology":null},{"url":"/paper/partially-relevant-video-retrieval","slug":"partially-relevant-video-retrieval","title":"Partially Relevant Video Retrieval","date":"2022-08-26","arxiv_id":"2208.12510","repositories_listed":1,"syntology":null},{"url":"/paper/diverse-video-captioning-by-adaptive-spatio","slug":"diverse-video-captioning-by-adaptive-spatio","title":"Diverse Video Captioning by Adaptive Spatio-temporal Attention","date":"2022-08-19","arxiv_id":"2208.09266","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-video-captioning-with-evolving","slug":"zero-shot-video-captioning-with-evolving","title":"Zero-Shot Video Captioning with Evolving Pseudo-Tokens","date":"2022-07-22","arxiv_id":"2207.11100","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-event-detection-and-captioning-as","slug":"unifying-event-detection-and-captioning-as","title":"Unifying Event Detection and Captioning as Sequence Generation via Pre-Training","date":"2022-07-18","arxiv_id":"2207.08625","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":4,"n_ran_checked":5,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/unifying-event-detection-and-captioning-as#ran","syntology_url":"https://syntology.ai/paper/2207.08625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.08625"}},"official":{"repos":["qiqang/uedvc"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-stream-transformer-for-generic-event","slug":"dual-stream-transformer-for-generic-event","title":"Dual-Stream Transformer for Generic Event Boundary Captioning","date":"2022-07-07","arxiv_id":"2207.03038","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-surgical-captioning-end-to-end","slug":"rethinking-surgical-captioning-end-to-end","title":"Rethinking Surgical Captioning: End-to-End Window-Based MLP Transformer Using Patches","date":"2022-06-30","arxiv_id":"2207.00113","repositories_listed":1,"syntology":null},{"url":"/paper/vlcap-vision-language-with-contrastive","slug":"vlcap-vision-language-with-contrastive","title":"VLCap: Vision-Language with Contrastive Learning for Coherent Video Paragraph Captioning","date":"2022-06-26","arxiv_id":"2206.12972","repositories_listed":1,"syntology":null},{"url":"/paper/lavender-unifying-video-language","slug":"lavender-unifying-video-language","title":"LAVENDER: Unifying Video-Language Understanding as Masked Language Modeling","date":"2022-06-14","arxiv_id":"2206.07160","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/lavender-unifying-video-language#ran","syntology_url":"https://syntology.ai/paper/2206.07160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07160"}},"official":{"repos":["microsoft/lavender"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/uni-perceiver-moe-learning-sparse-generalist","slug":"uni-perceiver-moe-learning-sparse-generalist","title":"Uni-Perceiver-MoE: Learning Sparse Generalist Models with Conditional MoEs","date":"2022-06-09","arxiv_id":"2206.04674","repositories_listed":1,"syntology":null},{"url":"/paper/git-a-generative-image-to-text-transformer","slug":"git-a-generative-image-to-text-transformer","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","date":"2022-05-27","arxiv_id":"2205.14100","repositories_listed":1,"syntology":{"n":21,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/git-a-generative-image-to-text-transformer#ran","syntology_url":"https://syntology.ai/paper/2205.14100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14100"}},"official":{"repos":["microsoft/GenerativeImage2Text"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/gl-rg-global-local-representation-granularity","slug":"gl-rg-global-local-representation-granularity","title":"GL-RG: Global-Local Representation Granularity for Video Captioning","date":"2022-05-22","arxiv_id":"2205.10706","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/gl-rg-global-local-representation-granularity#ran","syntology_url":"https://syntology.ai/paper/2205.10706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10706"}},"official":{"repos":["ylqi/gl-rg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-with-image-descriptors-are","slug":"language-models-with-image-descriptors-are","title":"Language Models with Image Descriptors are Strong Few-Shot Video-Language Learners","date":"2022-05-22","arxiv_id":"2205.10747","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-with-image-descriptors-are#ran","syntology_url":"https://syntology.ai/paper/2205.10747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10747"}},"official":{"repos":["mikewangwzhl/vidil"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/support-set-based-multi-modal-representation","slug":"support-set-based-multi-modal-representation","title":"Support-set based Multi-modal Representation Enhancement for Video Captioning","date":"2022-05-19","arxiv_id":"2205.09307","repositories_listed":1,"syntology":null},{"url":"/paper/tragedy-plus-time-capturing-unintended-human","slug":"tragedy-plus-time-capturing-unintended-human","title":"Tragedy Plus Time: Capturing Unintended Human Activities from Weakly-labeled Videos","date":"2022-04-28","arxiv_id":"2204.13548","repositories_listed":1,"syntology":null},{"url":"/paper/bertha-video-captioning-evaluation-via","slug":"bertha-video-captioning-evaluation-via","title":"BERTHA: Video Captioning Evaluation Via Transfer-Learned Human Assessment","date":"2022-01-25","arxiv_id":"2201.10243","repositories_listed":1,"syntology":null},{"url":"/paper/tell-me-what-you-see-a-zero-shot-action","slug":"tell-me-what-you-see-a-zero-shot-action","title":"Tell me what you see: A zero-shot action recognition method based on natural language descriptions","date":"2021-12-18","arxiv_id":"2112.09976","repositories_listed":1,"syntology":null},{"url":"/paper/dense-video-captioning-using-unsupervised","slug":"dense-video-captioning-using-unsupervised","title":"Dense Video Captioning Using Unsupervised Semantic Information","date":"2021-12-15","arxiv_id":"2112.08455","repositories_listed":1,"syntology":null},{"url":"/paper/controllable-video-captioning-with-an","slug":"controllable-video-captioning-with-an","title":"Controllable Video Captioning with an Exemplar Sentence","date":"2021-12-02","arxiv_id":"2112.01073","repositories_listed":1,"syntology":null},{"url":"/paper/syntax-customized-video-captioning-by","slug":"syntax-customized-video-captioning-by","title":"Syntax Customized Video Captioning by Imitating Exemplar Sentences","date":"2021-12-02","arxiv_id":"2112.01062","repositories_listed":1,"syntology":null},{"url":"/paper/clip-meets-video-captioners-attribute-aware","slug":"clip-meets-video-captioners-attribute-aware","title":"CLIP Meets Video Captioning: Concept-Aware Representation Learning Does Matter","date":"2021-11-30","arxiv_id":"2111.15162","repositories_listed":1,"syntology":null},{"url":"/paper/swinbert-end-to-end-transformers-with-sparse","slug":"swinbert-end-to-end-transformers-with-sparse","title":"SwinBERT: End-to-End Transformers with Sparse Attention for Video Captioning","date":"2021-11-25","arxiv_id":"2111.13196","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swinbert-end-to-end-transformers-with-sparse#ran","syntology_url":"https://syntology.ai/paper/2111.13196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.13196"}},"official":{"repos":["microsoft/swinbert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-modular-network-for-video","slug":"hierarchical-modular-network-for-video","title":"Hierarchical Modular Network for Video Captioning","date":"2021-11-24","arxiv_id":"2111.12476","repositories_listed":1,"syntology":null},{"url":"/paper/emscore-evaluating-video-captioning-via","slug":"emscore-evaluating-video-captioning-via","title":"EMScore: Evaluating Video Captioning via Coarse-Grained and Fine-Grained Embedding Matching","date":"2021-11-17","arxiv_id":"2111.08919","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emscore-evaluating-video-captioning-via#ran","syntology_url":"https://syntology.ai/paper/2111.08919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08919"}},"official":{"repos":["shiyaya/emscore"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/co-segmentation-inspired-attention-module-for","slug":"co-segmentation-inspired-attention-module-for","title":"Co-segmentation Inspired Attention Module for Video-based Computer Vision Tasks","date":"2021-11-14","arxiv_id":"2111.07370","repositories_listed":1,"syntology":null},{"url":"/paper/osvidcap-a-framework-for-the-simultaneous","slug":"osvidcap-a-framework-for-the-simultaneous","title":"OSVidCap: A Framework for the Simultaneous Recognition and Description of Concurrent Actions in Videos in an Open-Set Scenario","date":"2021-09-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sensor-augmented-egocentric-video-captioning","slug":"sensor-augmented-egocentric-video-captioning","title":"Sensor-Augmented Egocentric-Video Captioning with Dynamic Modal Attention","date":"2021-09-07","arxiv_id":"2109.02955","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-graph-with-meta-concepts-for","slug":"cross-modal-graph-with-meta-concepts-for","title":"Cross-Modal Graph with Meta Concepts for Video Captioning","date":"2021-08-14","arxiv_id":"2108.06458","repositories_listed":1,"syntology":null},{"url":"/paper/discriminative-latent-semantic-graph-for","slug":"discriminative-latent-semantic-graph-for","title":"Discriminative Latent Semantic Graph for Video Captioning","date":"2021-08-08","arxiv_id":"2108.03662","repositories_listed":1,"syntology":null},{"url":"/paper/value-a-multi-task-benchmark-for-video-and","slug":"value-a-multi-task-benchmark-for-video-and","title":"VALUE: A Multi-Task Benchmark for Video-and-Language Understanding Evaluation","date":"2021-06-08","arxiv_id":"2106.04632","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/value-a-multi-task-benchmark-for-video-and#ran","syntology_url":"https://syntology.ai/paper/2106.04632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04632"}},"official":{"repos":["VALUE-Leaderboard/StarterCode"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/decembert-learning-from-noisy-instructional","slug":"decembert-learning-from-noisy-instructional","title":"DeCEMBERT: Learning from Noisy Instructional Videos via Dense Captions and Entropy Minimization","date":"2021-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/improving-generation-and-evaluation-of-visual","slug":"improving-generation-and-evaluation-of-visual","title":"Improving Generation and Evaluation of Visual Stories via Semantic Consistency","date":"2021-05-20","arxiv_id":"2105.10026","repositories_listed":1,"syntology":{"n":14,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/improving-generation-and-evaluation-of-visual#ran","syntology_url":"https://syntology.ai/paper/2105.10026","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.10026"}},"official":{"repos":["adymaharana/StoryViz"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/vlm-task-agnostic-video-language-model-pre","slug":"vlm-task-agnostic-video-language-model-pre","title":"VLM: Task-agnostic Video-Language Model Pre-training for Video Understanding","date":"2021-05-20","arxiv_id":"2105.09996","repositories_listed":1,"syntology":null},{"url":"/paper/fill-in-the-blank-as-a-challenging-video","slug":"fill-in-the-blank-as-a-challenging-video","title":"FIBER: Fill-in-the-Blanks as a Challenging Video Understanding Evaluation Framework","date":"2021-04-09","arxiv_id":"2104.04182","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-vision-and-language-from-the-video","slug":"bridging-vision-and-language-from-the-video","title":"A Comprehensive Review of the Video-to-Text Problem","date":"2021-03-27","arxiv_id":"2103.14785","repositories_listed":1,"syntology":null},{"url":"/paper/annotation-cleaning-for-the-msr-video-to-text","slug":"annotation-cleaning-for-the-msr-video-to-text","title":"The MSR-Video to Text Dataset with Clean Annotations","date":"2021-02-12","arxiv_id":"2102.06448","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-grouping-network-for-video","slug":"semantic-grouping-network-for-video","title":"Semantic Grouping Network for Video Captioning","date":"2021-02-01","arxiv_id":"2102.00831","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-based-encoder","slug":"a-reinforcement-learning-based-encoder","title":"A Reinforcement Learning Based Encoder-Decoder Framework for Learning Stock Trading Rules","date":"2021-01-08","arxiv_id":"2101.03867","repositories_listed":1,"syntology":null},{"url":"/paper/tsp-temporally-sensitive-pretraining-of-video","slug":"tsp-temporally-sensitive-pretraining-of-video","title":"TSP: Temporally-Sensitive Pretraining of Video Encoders for Localization Tasks","date":"2020-11-23","arxiv_id":"2011.11479","repositories_listed":1,"syntology":null},{"url":"/paper/neuro-symbolic-representations-for-video","slug":"neuro-symbolic-representations-for-video","title":"Neuro-Symbolic Representations for Video Captioning: A Case for Leveraging Inductive Biases for Vision and Language","date":"2020-11-18","arxiv_id":"2011.09530","repositories_listed":1,"syntology":null},{"url":"/paper/actbert-learning-global-local-video-text-1","slug":"actbert-learning-global-local-video-text-1","title":"ActBERT: Learning Global-Local Video-Text Representations","date":"2020-11-14","arxiv_id":"2011.07231","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-pretraining-for-dense-video","slug":"multimodal-pretraining-for-dense-video","title":"Multimodal Pretraining for Dense Video Captioning","date":"2020-11-10","arxiv_id":"2011.11760","repositories_listed":1,"syntology":null},{"url":"/paper/coot-cooperative-hierarchical-transformer-for","slug":"coot-cooperative-hierarchical-transformer-for","title":"COOT: Cooperative Hierarchical Transformer for Video-Text Representation Learning","date":"2020-11-01","arxiv_id":"2011.00597","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coot-cooperative-hierarchical-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2011.00597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00597"}},"official":{"repos":["gingsi/coot-videotext"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-understanding-based-on-human-action-and","slug":"video-understanding-based-on-human-action-and","title":"Improved Actor Relation Graph based Group Activity Recognition","date":"2020-10-24","arxiv_id":"2010.12968","repositories_listed":1,"syntology":null},{"url":"/paper/semantically-sensible-video-captioning-ssvc","slug":"semantically-sensible-video-captioning-ssvc","title":"Video captioning with stacked attention and semantic hard pull","date":"2020-09-15","arxiv_id":"2009.07335","repositories_listed":1,"syntology":null},{"url":"/paper/poet-product-oriented-video-captioner-for-e","slug":"poet-product-oriented-video-captioner-for-e","title":"Poet: Product-oriented Video Captioner for E-commerce","date":"2020-08-16","arxiv_id":"2008.06880","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-generate-grounded-visual-captions","slug":"learning-to-generate-grounded-visual-captions","title":"Learning to Generate Grounded Visual Captions without Localization Supervision","date":"2020-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/soda-story-oriented-dense-video-captioning","slug":"soda-story-oriented-dense-video-captioning","title":"SODA: Story Oriented Dense Video Captioning Evaluation Framework","date":"2020-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-discretely-compose-reasoning","slug":"learning-to-discretely-compose-reasoning","title":"Learning to Discretely Compose Reasoning Module Networks for Video Captioning","date":"2020-07-17","arxiv_id":"2007.09049","repositories_listed":1,"syntology":null},{"url":"/paper/comprehensive-information-integration","slug":"comprehensive-information-integration","title":"Comprehensive Information Integration Modeling Framework for Video Titling","date":"2020-06-24","arxiv_id":"2006.13608","repositories_listed":1,"syntology":null},{"url":"/paper/dense-captioning-events-in-videos-sysu","slug":"dense-captioning-events-in-videos-sysu","title":"Dense-Captioning Events in Videos: SYSU Submission to ActivityNet Challenge 2020","date":"2020-06-21","arxiv_id":"2006.11693","repositories_listed":1,"syntology":null},{"url":"/paper/video-moment-localization-using-object","slug":"video-moment-localization-using-object","title":"Video Moment Localization using Object Evidence and Reverse Captioning","date":"2020-06-18","arxiv_id":"2006.10260","repositories_listed":1,"syntology":null},{"url":"/paper/screencast-tutorial-video-understanding","slug":"screencast-tutorial-video-understanding","title":"Screencast Tutorial Video Understanding","date":"2020-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/syntax-aware-action-targeting-for-video","slug":"syntax-aware-action-targeting-for-video","title":"Syntax-Aware Action Targeting for Video Captioning","date":"2020-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mart-memory-augmented-recurrent-transformer","slug":"mart-memory-augmented-recurrent-transformer","title":"MART: Memory-Augmented Recurrent Transformer for Coherent Video Paragraph Captioning","date":"2020-05-11","arxiv_id":"2005.05402","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/mart-memory-augmented-recurrent-transformer#ran","syntology_url":"https://syntology.ai/paper/2005.05402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.05402"}},"official":{"repos":["jayleicn/recurrent-transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-benchmark-for-structured-procedural","slug":"a-benchmark-for-structured-procedural","title":"A Benchmark for Structured Procedural Knowledge Extraction from Cooking Videos","date":"2020-05-02","arxiv_id":"2005.00706","repositories_listed":1,"syntology":null},{"url":"/paper/delving-deeper-into-the-decoder-for-video","slug":"delving-deeper-into-the-decoder-for-video","title":"Delving Deeper into the Decoder for Video Captioning","date":"2020-01-16","arxiv_id":"2001.05614","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/delving-deeper-into-the-decoder-for-video#ran","syntology_url":"https://syntology.ai/paper/2001.05614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.05614"}},"official":{"repos":["WingsBrokenAngel/delving-deeper-into-the-decoder-for-video-captioning"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/meaning-guided-video-captioning","slug":"meaning-guided-video-captioning","title":"Meaning guided video captioning","date":"2019-12-12","arxiv_id":"1912.05730","repositories_listed":1,"syntology":null},{"url":"/paper/non-autoregressive-video-captioning-with","slug":"non-autoregressive-video-captioning-with","title":"Non-Autoregressive Coarse-to-Fine Video Captioning","date":"2019-11-27","arxiv_id":"1911.12018","repositories_listed":1,"syntology":null},{"url":"/paper/multi-attention-networks-for-temporal","slug":"multi-attention-networks-for-temporal","title":"Multi-attention Networks for Temporal Localization of Video-level Labels","date":"2019-11-15","arxiv_id":"1911.06866","repositories_listed":1,"syntology":null},{"url":"/paper/contcap-a-comprehensive-framework-for","slug":"contcap-a-comprehensive-framework-for","title":"ContCap: A scalable framework for continual image captioning","date":"2019-09-19","arxiv_id":"1909.08745","repositories_listed":1,"syntology":null},{"url":"/paper/controllable-video-captioning-with-pos","slug":"controllable-video-captioning-with-pos","title":"Controllable Video Captioning with POS Sequence Guidance Based on Gated Fusion Network","date":"2019-08-27","arxiv_id":"1908.10072","repositories_listed":1,"syntology":null},{"url":"/paper/continual-and-multi-task-architecture-search","slug":"continual-and-multi-task-architecture-search","title":"Continual and Multi-Task Architecture Search","date":"2019-06-12","arxiv_id":"1906.05226","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/continual-and-multi-task-architecture-search#ran","syntology_url":"https://syntology.ai/paper/1906.05226","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.05226"}},"official":{"repos":["ramakanth-pasunuru/CAS-MAS"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-neural-interactive-predictive-system-for","slug":"a-neural-interactive-predictive-system-for","title":"A Neural, Interactive-predictive System for Multimodal Sequence to Sequence Tasks","date":"2019-05-20","arxiv_id":"1905.08181","repositories_listed":1,"syntology":null},{"url":"/paper/memory-attended-recurrent-network-for-video","slug":"memory-attended-recurrent-network-for-video","title":"Memory-Attended Recurrent Network for Video Captioning","date":"2019-05-10","arxiv_id":"1905.03966","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-deformable-convolutional-encoder","slug":"temporal-deformable-convolutional-encoder","title":"Temporal Deformable Convolutional Encoder-Decoder Networks for Video Captioning","date":"2019-05-03","arxiv_id":"1905.01077","repositories_listed":1,"syntology":null},{"url":"/paper/holistic-large-scale-video-understanding","slug":"holistic-large-scale-video-understanding","title":"Large Scale Holistic Video Understanding","date":"2019-04-25","arxiv_id":"1904.11451","repositories_listed":1,"syntology":null},{"url":"/paper/membership-inference-attacks-on-sequence-to","slug":"membership-inference-attacks-on-sequence-to","title":"Membership Inference Attacks on Sequence-to-Sequence Models: Is My Data In Your Machine Translation System?","date":"2019-04-11","arxiv_id":"1904.05506","repositories_listed":1,"syntology":null},{"url":"/paper/streamlined-dense-video-captioning","slug":"streamlined-dense-video-captioning","title":"Streamlined Dense Video Captioning","date":"2019-04-08","arxiv_id":"1904.03870","repositories_listed":1,"syntology":null},{"url":"/paper/m-vad-names-a-dataset-for-video-captioning","slug":"m-vad-names-a-dataset-for-video-captioning","title":"M-VAD Names: a Dataset for Video Captioning with Naming","date":"2019-03-04","arxiv_id":"1903.01489","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-and-hierarchical-modeling-of","slug":"cross-modal-and-hierarchical-modeling-of","title":"Cross-Modal and Hierarchical Modeling of Video and Text","date":"2018-10-16","arxiv_id":"1810.07212","repositories_listed":1,"syntology":null},{"url":"/paper/mtle-a-multitask-learning-encoder-of-visual","slug":"mtle-a-multitask-learning-encoder-of-visual","title":"MTLE: A Multitask Learning Encoder of Visual Feature Representations for Video and Movie Description","date":"2018-09-19","arxiv_id":"1809.07257","repositories_listed":1,"syntology":null},{"url":"/paper/nmt-keras-a-very-flexible-toolkit-with-a","slug":"nmt-keras-a-very-flexible-toolkit-with-a","title":"NMT-Keras: a Very Flexible Toolkit with a Focus on Interactive NMT and Online Learning","date":"2018-07-09","arxiv_id":"1807.03096","repositories_listed":1,"syntology":null},{"url":"/paper/watch-listen-and-describe-globally-and","slug":"watch-listen-and-describe-globally-and","title":"Watch, Listen, and Describe: Globally and Locally Aligned Cross-Modal Attentions for Video Captioning","date":"2018-04-15","arxiv_id":"1804.05448","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-dense-video-captioning-with-masked","slug":"end-to-end-dense-video-captioning-with-masked","title":"End-to-End Dense Video Captioning with Masked Transformer","date":"2018-04-03","arxiv_id":"1804.00819","repositories_listed":1,"syntology":null},{"url":"/paper/bidirectional-attentive-fusion-with-context","slug":"bidirectional-attentive-fusion-with-context","title":"Bidirectional Attentive Fusion with Context Gating for Dense Video Captioning","date":"2018-03-31","arxiv_id":"1804.00100","repositories_listed":1,"syntology":null}],"record_sha256":"5ee41112681f8928d9d1a897b566f2e3759d3f8752a078aa902216d5e83349e3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}