{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-retrieval/papers/2","list_of":"/task/video-retrieval","task":"Video Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":5,"rows_per_page":100,"rows":[101,200],"of":486,"counts":{"archive_papers_tagged":486,"with_a_code_link":255,"where_syntology_ran_a_sample":95,"not_listed_spam_title":0,"listed":486,"listed_where_code_ran":95,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":76,"every_run_a_failure_of_syntologys_instrument":19,"listed_with_a_run_with_no_instrument_failure":76,"listed_every_run_a_failure_of_syntologys_instrument":19,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-retrieval","prev":"/task/video-retrieval","next":"/task/video-retrieval/papers/3","papers":[{"url":"/paper/howtocaption-prompting-llms-to-transform","slug":"howtocaption-prompting-llms-to-transform","title":"HowToCaption: Prompting LLMs to Transform Video Annotations at Scale","date":"2023-10-07","arxiv_id":"2310.04900","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/howtocaption-prompting-llms-to-transform#ran","syntology_url":"https://syntology.ai/paper/2310.04900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04900"}},"official":{"repos":["ninatu/howtocaption"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/prototype-based-aleatoric-uncertainty-1","slug":"prototype-based-aleatoric-uncertainty-1","title":"Prototype-based Aleatoric Uncertainty Quantification for Cross-modal Retrieval","date":"2023-09-29","arxiv_id":"2309.17093","repositories_listed":1,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":10,"n_instrument":7,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":7,"n_pointer_only":8,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 1 violated, 7 with no contract checked; 7 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/prototype-based-aleatoric-uncertainty-1#ran","syntology_url":"https://syntology.ai/paper/2309.17093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17093"}},"official":{"repos":["leolee99/pau"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-modal-attention-enhanced-text-video","slug":"dual-modal-attention-enhanced-text-video","title":"Dual-Modal Attention-Enhanced Text-Video Retrieval with Triplet Partial Margin Contrastive Learning","date":"2023-09-20","arxiv_id":"2309.11082","repositories_listed":1,"syntology":null},{"url":"/paper/unified-coarse-to-fine-alignment-for-video","slug":"unified-coarse-to-fine-alignment-for-video","title":"Unified Coarse-to-Fine Alignment for Video-Text Retrieval","date":"2023-09-18","arxiv_id":"2309.10091","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unified-coarse-to-fine-alignment-for-video#ran","syntology_url":"https://syntology.ai/paper/2309.10091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.10091"}},"official":{"repos":["ziyang412/ucofia"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/in-style-bridging-text-and-uncurated-videos","slug":"in-style-bridging-text-and-uncurated-videos","title":"In-Style: Bridging Text and Uncurated Videos with Style Transfer for Text-Video Retrieval","date":"2023-09-16","arxiv_id":"2309.08928","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-resolution-compression-and","slug":"differentiable-resolution-compression-and","title":"Differentiable Resolution Compression and Alignment for Efficient Video Classification and Retrieval","date":"2023-09-15","arxiv_id":"2309.08167","repositories_listed":1,"syntology":null},{"url":"/paper/language-conditioned-change-point-detection","slug":"language-conditioned-change-point-detection","title":"Language-Conditioned Change-point Detection to Identify Sub-Tasks in Robotics Domains","date":"2023-09-01","arxiv_id":"2309.00743","repositories_listed":1,"syntology":null},{"url":"/paper/covr-learning-composed-video-retrieval-from","slug":"covr-learning-composed-video-retrieval-from","title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","date":"2023-08-28","arxiv_id":"2308.14746","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/covr-learning-composed-video-retrieval-from#ran","syntology_url":"https://syntology.ai/paper/2308.14746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14746"}},"official":{"repos":["lucas-ventura/CoVR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/simple-baselines-for-interactive-video","slug":"simple-baselines-for-interactive-video","title":"Simple Baselines for Interactive Video Retrieval with Questions and Answers","date":"2023-08-21","arxiv_id":"2308.10402","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/simple-baselines-for-interactive-video#ran","syntology_url":"https://syntology.ai/paper/2308.10402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10402"}},"official":{"repos":["kevinliang888/ivr-qa-baselines"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/prompt-switch-efficient-clip-adaptation-for","slug":"prompt-switch-efficient-clip-adaptation-for","title":"Prompt Switch: Efficient CLIP Adaptation for Text-Video Retrieval","date":"2023-08-15","arxiv_id":"2308.07648","repositories_listed":1,"syntology":{"n":16,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":12,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":16,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 12 unverified","sample_list":"/paper/prompt-switch-efficient-clip-adaptation-for#ran","syntology_url":"https://syntology.ai/paper/2308.07648","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07648"}},"official":{"repos":["bladewaltz1/promptswitch"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":12,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-video-anomaly-retrieval-from-video","slug":"towards-video-anomaly-retrieval-from-video","title":"Towards Video Anomaly Retrieval from Video Anomaly Detection: New Benchmarks and Model","date":"2023-07-24","arxiv_id":"2307.12545","repositories_listed":1,"syntology":null},{"url":"/paper/animate-a-story-storytelling-with-retrieval","slug":"animate-a-story-storytelling-with-retrieval","title":"Animate-A-Story: Storytelling with Retrieval-Augmented Video Generation","date":"2023-07-13","arxiv_id":"2307.06940","repositories_listed":1,"syntology":null},{"url":"/paper/internvid-a-large-scale-video-text-dataset","slug":"internvid-a-large-scale-video-text-dataset","title":"InternVid: A Large-scale Video-Text Dataset for Multimodal Understanding and Generation","date":"2023-07-13","arxiv_id":"2307.06942","repositories_listed":1,"syntology":null},{"url":"/paper/an-overview-on-the-evaluated-video-retrieval","slug":"an-overview-on-the-evaluated-video-retrieval","title":"An overview on the evaluated video retrieval tasks at TRECVID 2022","date":"2023-06-22","arxiv_id":"2306.13118","repositories_listed":1,"syntology":null},{"url":"/paper/msvd-indonesian-a-benchmark-for-multimodal","slug":"msvd-indonesian-a-benchmark-for-multimodal","title":"MSVD-Indonesian: A Benchmark for Multimodal Video-Text Tasks in Indonesian","date":"2023-06-20","arxiv_id":"2306.11341","repositories_listed":1,"syntology":null},{"url":"/paper/cosa-concatenated-sample-pretrained-vision","slug":"cosa-concatenated-sample-pretrained-vision","title":"COSA: Concatenated Sample Pretrained Vision-Language Foundation Model","date":"2023-06-15","arxiv_id":"2306.09085","repositories_listed":1,"syntology":null},{"url":"/paper/a-large-cross-modal-video-retrieval-dataset","slug":"a-large-cross-modal-video-retrieval-dataset","title":"A Large Cross-Modal Video Retrieval Dataset with Reading Comprehension","date":"2023-05-05","arxiv_id":"2305.03347","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-large-cross-modal-video-retrieval-dataset#ran","syntology_url":"https://syntology.ai/paper/2305.03347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.03347"}},"official":{"repos":["callsys/textvr"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/valor-vision-audio-language-omni-perception","slug":"valor-vision-audio-language-omni-perception","title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","date":"2023-04-17","arxiv_id":"2304.08345","repositories_listed":1,"syntology":null},{"url":"/paper/robust-cross-modal-knowledge-distillation-for","slug":"robust-cross-modal-knowledge-distillation-for","title":"Robust Cross-Modal Knowledge Distillation for Unconstrained Videos","date":"2023-04-16","arxiv_id":"2304.07775","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-cross-modal-knowledge-distillation-for#ran","syntology_url":"https://syntology.ai/paper/2304.07775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.07775"}},"official":{"repos":["gewu-lab/cross-modal-distillation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-video-similarity-learning","slug":"self-supervised-video-similarity-learning","title":"Self-Supervised Video Similarity Learning","date":"2023-04-06","arxiv_id":"2304.03378","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-supervised-video-similarity-learning#ran","syntology_url":"https://syntology.ai/paper/2304.03378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.03378"}},"official":{"repos":["gkordo/s2vs"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-video-moment-retrieval-and-step","slug":"hierarchical-video-moment-retrieval-and-step","title":"Hierarchical Video-Moment Retrieval and Step-Captioning","date":"2023-03-29","arxiv_id":"2303.16406","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hierarchical-video-moment-retrieval-and-step#ran","syntology_url":"https://syntology.ai/paper/2303.16406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16406"}},"official":{"repos":["j-min/HiREST"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unmasked-teacher-towards-training-efficient","slug":"unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","arxiv_id":"2303.16058","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unmasked-teacher-towards-training-efficient#ran","syntology_url":"https://syntology.ai/paper/2303.16058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16058"}},"official":{"repos":["opengvlab/unmasked_teacher"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aligning-step-by-step-instructional-diagrams","slug":"aligning-step-by-step-instructional-diagrams","title":"Aligning Step-by-Step Instructional Diagrams to Video Demonstrations","date":"2023-03-24","arxiv_id":"2303.13800","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aligning-step-by-step-instructional-diagrams#ran","syntology_url":"https://syntology.ai/paper/2303.13800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13800"}},"official":{"repos":["DavidZhang73/AssemblyVideoManualAlignment"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dialogue-to-video-retrieval","slug":"dialogue-to-video-retrieval","title":"Dialogue-to-Video Retrieval","date":"2023-03-23","arxiv_id":"2303.16761","repositories_listed":1,"syntology":null},{"url":"/paper/meltr-meta-loss-transformer-for-learning-to","slug":"meltr-meta-loss-transformer-for-learning-to","title":"MELTR: Meta Loss Transformer for Learning to Fine-tune Video Foundation Models","date":"2023-03-23","arxiv_id":"2303.13009","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/meltr-meta-loss-transformer-for-learning-to#ran","syntology_url":"https://syntology.ai/paper/2303.13009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13009"}},"official":{"repos":["mlvlab/MELTR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vvs-video-to-video-retrieval-with-irrelevant","slug":"vvs-video-to-video-retrieval-with-irrelevant","title":"VVS: Video-to-Video Retrieval with Irrelevant Frame Suppression","date":"2023-03-15","arxiv_id":"2303.08906","repositories_listed":1,"syntology":null},{"url":"/paper/accommodating-audio-modality-in-clip-for","slug":"accommodating-audio-modality-in-clip-for","title":"Accommodating Audio Modality in CLIP for Multimodal Processing","date":"2023-03-12","arxiv_id":"2303.06591","repositories_listed":1,"syntology":null},{"url":"/paper/video-text-retrieval-by-supervised-multi","slug":"video-text-retrieval-by-supervised-multi","title":"Video-Text Retrieval by Supervised Sparse Multi-Grained Learning","date":"2023-02-19","arxiv_id":"2302.09473","repositories_listed":1,"syntology":null},{"url":"/paper/is-multi-modal-vision-supervision-beneficial","slug":"is-multi-modal-vision-supervision-beneficial","title":"Is Multimodal Vision Supervision Beneficial to Language?","date":"2023-02-10","arxiv_id":"2302.05016","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-end-to-end-video-question-answering","slug":"efficient-end-to-end-video-question-answering","title":"Efficient End-to-End Video Question Answering with Pyramidal Multimodal Transformer","date":"2023-02-04","arxiv_id":"2302.02136","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-temporal-modeling-for-clip-based","slug":"revisiting-temporal-modeling-for-clip-based","title":"Revisiting Temporal Modeling for CLIP-based Image-to-Video Knowledge Transferring","date":"2023-01-26","arxiv_id":"2301.11116","repositories_listed":1,"syntology":null},{"url":"/paper/zorro-the-masked-multimodal-transformer","slug":"zorro-the-masked-multimodal-transformer","title":"Zorro: the masked multimodal transformer","date":"2023-01-23","arxiv_id":"2301.09595","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zorro-the-masked-multimodal-transformer#ran","syntology_url":"https://syntology.ai/paper/2301.09595","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.09595"}},"official":null}},{"url":"/paper/uatvr-uncertainty-adaptive-text-video","slug":"uatvr-uncertainty-adaptive-text-video","title":"UATVR: Uncertainty-Adaptive Text-Video Retrieval","date":"2023-01-16","arxiv_id":"2301.06309","repositories_listed":1,"syntology":null},{"url":"/paper/dual-learning-with-dynamic-knowledge","slug":"dual-learning-with-dynamic-knowledge","title":"Dual Learning with Dynamic Knowledge Distillation for Partially Relevant Video Retrieval","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/exploring-temporal-concurrency-for-video","slug":"exploring-temporal-concurrency-for-video","title":"Exploring Temporal Concurrency for Video-Language Representation Learning","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/progressive-spatio-temporal-prototype","slug":"progressive-spatio-temporal-prototype","title":"Progressive Spatio-Temporal Prototype Matching for Text-Video Retrieval","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/tempclr-temporal-alignment-representation","slug":"tempclr-temporal-alignment-representation","title":"TempCLR: Temporal Alignment Representation with Contrastive Learning","date":"2022-12-28","arxiv_id":"2212.13738","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tempclr-temporal-alignment-representation#ran","syntology_url":"https://syntology.ai/paper/2212.13738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.13738"}},"official":{"repos":["yyuncong/tempclr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/you-were-saying-spoken-language-in-the-v3c","slug":"you-were-saying-spoken-language-in-the-v3c","title":"You were saying? - Spoken Language in the V3C Dataset","date":"2022-12-15","arxiv_id":"2212.07835","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-explainable-video-representation","slug":"contextual-explainable-video-representation","title":"Contextual Explainable Video Representation: Human Perception-based Understanding","date":"2022-12-12","arxiv_id":"2212.06206","repositories_listed":1,"syntology":null},{"url":"/paper/vindlu-a-recipe-for-effective-video-and","slug":"vindlu-a-recipe-for-effective-video-and","title":"VindLU: A Recipe for Effective Video-and-Language Pretraining","date":"2022-12-09","arxiv_id":"2212.05051","repositories_listed":1,"syntology":null},{"url":"/paper/normalized-contrastive-learning-for-text","slug":"normalized-contrastive-learning-for-text","title":"Normalized Contrastive Learning for Text-Video Retrieval","date":"2022-11-30","arxiv_id":"2212.11790","repositories_listed":1,"syntology":null},{"url":"/paper/vop-text-video-co-operative-prompt-tuning-for","slug":"vop-text-video-co-operative-prompt-tuning-for","title":"VoP: Text-Video Co-operative Prompt Tuning for Cross-Modal Retrieval","date":"2022-11-23","arxiv_id":"2211.12764","repositories_listed":1,"syntology":null},{"url":"/paper/are-all-combinations-equal-combining-textual","slug":"are-all-combinations-equal-combining-textual","title":"Are All Combinations Equal? Combining Textual and Visual Features with Multiple Space Learning for Text-Based Video Retrieval","date":"2022-11-21","arxiv_id":"2211.11351","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-masked-autoencoders-for-self","slug":"contrastive-masked-autoencoders-for-self","title":"Contrastive Masked Autoencoders for Self-Supervised Video Hashing","date":"2022-11-21","arxiv_id":"2211.11210","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-adapter-for-text-video-retrieval","slug":"cross-modal-adapter-for-text-video-retrieval","title":"Cross-Modal Adapter for Text-Video Retrieval","date":"2022-11-17","arxiv_id":"2211.09623","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":10,"n_pointer_only":6,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 2 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-modal-adapter-for-text-video-retrieval#ran","syntology_url":"https://syntology.ai/paper/2211.09623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09623"}},"official":{"repos":["leaplabthu/cross-modal-adapter"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/3d-csl-self-supervised-3d-context-similarity","slug":"3d-csl-self-supervised-3d-context-similarity","title":"3D-CSL: self-supervised 3D context similarity learning for Near-Duplicate Video Retrieval","date":"2022-11-10","arxiv_id":"2211.05352","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-cross-modal-video-retrieval-with","slug":"efficient-cross-modal-video-retrieval-with","title":"Efficient Cross-Modal Video Retrieval with Meta-Optimized Frames","date":"2022-10-16","arxiv_id":"2210.08452","repositories_listed":1,"syntology":null},{"url":"/paper/rap-redundancy-aware-video-language-pre","slug":"rap-redundancy-aware-video-language-pre","title":"RaP: Redundancy-aware Video-language Pre-training for Text-Video Retrieval","date":"2022-10-13","arxiv_id":"2210.06881","repositories_listed":1,"syntology":null},{"url":"/paper/long-form-video-language-pre-training-with","slug":"long-form-video-language-pre-training-with","title":"Long-Form Video-Language Pre-Training with Multimodal Temporal Contrastive Learning","date":"2022-10-12","arxiv_id":"2210.06031","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/long-form-video-language-pre-training-with#ran","syntology_url":"https://syntology.ai/paper/2210.06031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06031"}},"official":{"repos":["microsoft/xpretrain"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-locate-visual-answer-in-video","slug":"learning-to-locate-visual-answer-in-video","title":"Learning to Locate Visual Answer in Video Corpus Using Question","date":"2022-10-11","arxiv_id":"2210.05423","repositories_listed":1,"syntology":null},{"url":"/paper/contra-con-text-tra-nsformer-for-cross-modal","slug":"contra-con-text-tra-nsformer-for-cross-modal","title":"ConTra: (Con)text (Tra)nsformer for Cross-Modal Video Retrieval","date":"2022-10-09","arxiv_id":"2210.04341","repositories_listed":1,"syntology":null},{"url":"/paper/c2kd-cross-lingual-cross-modal-knowledge","slug":"c2kd-cross-lingual-cross-modal-knowledge","title":"C2KD: Cross-Lingual Cross-Modal Knowledge Distillation for Multilingual Text-Video Retrieval","date":"2022-10-07","arxiv_id":"2210.03625","repositories_listed":1,"syntology":null},{"url":"/paper/marine-video-kit-a-new-marine-video-dataset","slug":"marine-video-kit-a-new-marine-video-dataset","title":"Marine Video Kit: A New Marine Video Dataset for Content-based Analysis and Retrieval","date":"2022-09-23","arxiv_id":"2209.11518","repositories_listed":1,"syntology":null},{"url":"/paper/clip-vip-adapting-pre-trained-image-text","slug":"clip-vip-adapting-pre-trained-image-text","title":"CLIP-ViP: Adapting Pre-trained Image-Text Model to Video-Language Representation Alignment","date":"2022-09-14","arxiv_id":"2209.06430","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clip-vip-adapting-pre-trained-image-text#ran","syntology_url":"https://syntology.ai/paper/2209.06430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.06430"}},"official":{"repos":["microsoft/xpretrain"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-of-end-to-end-video","slug":"an-empirical-study-of-end-to-end-video","title":"An Empirical Study of End-to-End Video-Language Transformers with Masked Visual Modeling","date":"2022-09-04","arxiv_id":"2209.01540","repositories_listed":1,"syntology":null},{"url":"/paper/partially-relevant-video-retrieval","slug":"partially-relevant-video-retrieval","title":"Partially Relevant Video Retrieval","date":"2022-08-26","arxiv_id":"2208.12510","repositories_listed":1,"syntology":null},{"url":"/paper/a-feature-space-multimodal-data-augmentation","slug":"a-feature-space-multimodal-data-augmentation","title":"A Feature-space Multimodal Data Augmentation Technique for Text-video Retrieval","date":"2022-08-03","arxiv_id":"2208.02080","repositories_listed":1,"syntology":null},{"url":"/paper/locvtp-video-text-pre-training-for-temporal","slug":"locvtp-video-text-pre-training-for-temporal","title":"LocVTP: Video-Text Pre-training for Temporal Localization","date":"2022-07-21","arxiv_id":"2207.10362","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/locvtp-video-text-pre-training-for-temporal#ran","syntology_url":"https://syntology.ai/paper/2207.10362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.10362"}},"official":{"repos":["mengcaopku/locvtp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/goca-guided-online-cluster-assignment-for","slug":"goca-guided-online-cluster-assignment-for","title":"GOCA: Guided Online Cluster Assignment for Self-Supervised Video Representation Learning","date":"2022-07-20","arxiv_id":"2207.10158","repositories_listed":1,"syntology":null},{"url":"/paper/clover-towards-a-unified-video-language","slug":"clover-towards-a-unified-video-language","title":"Clover: Towards A Unified Video-Language Alignment and Fusion Model","date":"2022-07-16","arxiv_id":"2207.07885","repositories_listed":1,"syntology":null},{"url":"/paper/ts2-net-token-shift-and-selection-transformer","slug":"ts2-net-token-shift-and-selection-transformer","title":"TS2-Net: Token Shift and Selection Transformer for Text-Video Retrieval","date":"2022-07-16","arxiv_id":"2207.07852","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/ts2-net-token-shift-and-selection-transformer#ran","syntology_url":"https://syntology.ai/paper/2207.07852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.07852"}},"official":{"repos":["yuqi657/ts2_net"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-robustness-analysis-against","slug":"multi-modal-robustness-analysis-against","title":"Robustness Analysis of Video-Language Models Against Visual and Language Perturbations","date":"2022-07-05","arxiv_id":"2207.02159","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-semantic-role-contextualized-video","slug":"exploiting-semantic-role-contextualized-video","title":"Exploiting Semantic Role Contextualized Video Features for Multi-Instance Text-Video Retrieval EPIC-KITCHENS-100 Multi-Instance Retrieval Challenge 2022","date":"2022-06-29","arxiv_id":"2206.14381","repositories_listed":1,"syntology":null},{"url":"/paper/rome-role-aware-mixture-of-expert-transformer","slug":"rome-role-aware-mixture-of-expert-transformer","title":"RoME: Role-aware Mixture-of-Expert Transformer for Text-to-Video Retrieval","date":"2022-06-26","arxiv_id":"2206.12845","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-role-aware-correlation-transformer","slug":"semantic-role-aware-correlation-transformer","title":"Semantic Role Aware Correlation Transformer for Text to Video Retrieval","date":"2022-06-26","arxiv_id":"2206.12849","repositories_listed":1,"syntology":null},{"url":"/paper/slic-self-supervised-learning-with-iterative-1","slug":"slic-self-supervised-learning-with-iterative-1","title":"SLIC: Self-Supervised Learning with Iterative Clustering for Human Action Videos","date":"2022-06-25","arxiv_id":"2206.12534","repositories_listed":1,"syntology":null},{"url":"/paper/lavender-unifying-video-language","slug":"lavender-unifying-video-language","title":"LAVENDER: Unifying Video-Language Understanding as Masked Language Modeling","date":"2022-06-14","arxiv_id":"2206.07160","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/lavender-unifying-video-language#ran","syntology_url":"https://syntology.ai/paper/2206.07160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07160"}},"official":{"repos":["microsoft/lavender"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/revisiting-the-video-in-video-language","slug":"revisiting-the-video-in-video-language","title":"Revisiting the \"Video\" in Video-Language Understanding","date":"2022-06-03","arxiv_id":"2206.01720","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/revisiting-the-video-in-video-language#ran","syntology_url":"https://syntology.ai/paper/2206.01720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01720"}},"official":null}},{"url":"/paper/cross-architecture-self-supervised-video","slug":"cross-architecture-self-supervised-video","title":"Cross-Architecture Self-supervised Video Representation Learning","date":"2022-05-26","arxiv_id":"2205.13313","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/cross-architecture-self-supervised-video#ran","syntology_url":"https://syntology.ai/paper/2205.13313","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13313"}},"official":{"repos":["guoshengcv/cacl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-clip-hitchhiker-s-guide-to-long-video","slug":"a-clip-hitchhiker-s-guide-to-long-video","title":"A CLIP-Hitchhiker's Guide to Long Video Retrieval","date":"2022-05-17","arxiv_id":"2205.08508","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-retrieve-videos-by-asking","slug":"learning-to-retrieve-videos-by-asking","title":"Learning to Retrieve Videos by Asking Questions","date":"2022-05-11","arxiv_id":"2205.05739","repositories_listed":1,"syntology":null},{"url":"/paper/transrank-self-supervised-video","slug":"transrank-self-supervised-video","title":"TransRank: Self-supervised Video Representation Learning via Ranking-based Transformation Recognition","date":"2022-05-04","arxiv_id":"2205.02028","repositories_listed":1,"syntology":null},{"url":"/paper/centerclip-token-clustering-for-efficient","slug":"centerclip-token-clustering-for-efficient","title":"CenterCLIP: Token Clustering for Efficient Text-Video Retrieval","date":"2022-05-02","arxiv_id":"2205.00823","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/centerclip-token-clustering-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2205.00823","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00823"}},"official":{"repos":["mzhaoshuai/CenterCLIP"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/learn-to-understand-negation-in-video","slug":"learn-to-understand-negation-in-video","title":"Learn to Understand Negation in Video Retrieval","date":"2022-04-30","arxiv_id":"2205.00132","repositories_listed":1,"syntology":null},{"url":"/paper/relevance-based-margin-for-contrastively","slug":"relevance-based-margin-for-contrastively","title":"Relevance-based Margin for Contrastively-trained Video Retrieval Models","date":"2022-04-27","arxiv_id":"2204.13001","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/relevance-based-margin-for-contrastively#ran","syntology_url":"https://syntology.ai/paper/2204.13001","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.13001"}},"official":{"repos":["aranciokov/relevancemargin-icmr22"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/miles-visual-bert-pre-training-with-injected","slug":"miles-visual-bert-pre-training-with-injected","title":"MILES: Visual BERT Pre-training with Injected Language Semantics for Video-text Retrieval","date":"2022-04-26","arxiv_id":"2204.12408","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-temporal-cues-to-enhance-video","slug":"exploring-the-temporal-cues-to-enhance-video","title":"Exploring the Temporal Cues to Enhance Video Retrieval on Standardized CDVA","date":"2022-04-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/eclipse-efficient-long-range-video-retrieval","slug":"eclipse-efficient-long-range-video-retrieval","title":"ECLIPSE: Efficient Long-range Video Retrieval using Sight and Sound","date":"2022-04-06","arxiv_id":"2204.02874","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/eclipse-efficient-long-range-video-retrieval#ran","syntology_url":"https://syntology.ai/paper/2204.02874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02874"}},"official":{"repos":["GenjiB/ECLIPSE"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-alignment-networks-for-long-term","slug":"temporal-alignment-networks-for-long-term","title":"Temporal Alignment Networks for Long-term Video","date":"2022-04-06","arxiv_id":"2204.02968","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/temporal-alignment-networks-for-long-term#ran","syntology_url":"https://syntology.ai/paper/2204.02968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02968"}},"official":null}},{"url":"/paper/socratic-models-composing-zero-shot","slug":"socratic-models-composing-zero-shot","title":"Socratic Models: Composing Zero-Shot Multimodal Reasoning with Language","date":"2022-04-01","arxiv_id":"2204.00598","repositories_listed":1,"syntology":null},{"url":"/paper/x-pool-cross-modal-language-video-attention","slug":"x-pool-cross-modal-language-video-attention","title":"X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval","date":"2022-03-28","arxiv_id":"2203.15086","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/x-pool-cross-modal-language-video-attention#ran","syntology_url":"https://syntology.ai/paper/2203.15086","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15086"}},"official":{"repos":["layer6ai-labs/xpool"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/all-in-one-exploring-unified-video-language","slug":"all-in-one-exploring-unified-video-language","title":"All in One: Exploring Unified Video-Language Pre-training","date":"2022-03-14","arxiv_id":"2203.07303","repositories_listed":1,"syntology":null},{"url":"/paper/show-me-more-details-discovering-hierarchies","slug":"show-me-more-details-discovering-hierarchies","title":"Show Me More Details: Discovering Hierarchies of Procedures from Semi-structured Web Data","date":"2022-03-14","arxiv_id":"2203.07264","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/show-me-more-details-discovering-hierarchies#ran","syntology_url":"https://syntology.ai/paper/2203.07264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.07264"}},"official":{"repos":["shuyanzhou/wikihow_hierarchy"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-contrastive-quantization-for-efficient","slug":"hybrid-contrastive-quantization-for-efficient","title":"Hybrid Contrastive Quantization for Efficient Cross-View Video Retrieval","date":"2022-02-07","arxiv_id":"2202.03384","repositories_listed":1,"syntology":null},{"url":"/paper/reading-strategy-inspired-visual","slug":"reading-strategy-inspired-visual","title":"Reading-strategy Inspired Visual Representation Learning for Text-to-Video Retrieval","date":"2022-01-23","arxiv_id":"2201.09168","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-video-representation-learning-10","slug":"self-supervised-video-representation-learning-10","title":"Self-supervised Video Representation Learning with Cascade Positive Retrieval","date":"2022-01-20","arxiv_id":"2201.07989","repositories_listed":1,"syntology":null},{"url":"/paper/multi-query-video-retrieval","slug":"multi-query-video-retrieval","title":"Multi-Query Video Retrieval","date":"2022-01-10","arxiv_id":"2201.03639","repositories_listed":1,"syntology":null},{"url":"/paper/everything-at-once-multi-modal-fusion-1","slug":"everything-at-once-multi-modal-fusion-1","title":"Everything at Once - Multi-Modal Fusion Transformer for Video Retrieval","date":"2022-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-retrieval-with-querybank","slug":"cross-modal-retrieval-with-querybank","title":"Cross Modal Retrieval with Querybank Normalisation","date":"2021-12-23","arxiv_id":"2112.12777","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cross-modal-retrieval-with-querybank#ran","syntology_url":"https://syntology.ai/paper/2112.12777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.12777"}},"official":{"repos":["ioanacroi/qb-norm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/align-and-prompt-video-and-language-pre","slug":"align-and-prompt-video-and-language-pre","title":"Align and Prompt: Video-and-Language Pre-training with Entity Prompts","date":"2021-12-17","arxiv_id":"2112.09583","repositories_listed":1,"syntology":null},{"url":"/paper/everything-at-once-multi-modal-fusion","slug":"everything-at-once-multi-modal-fusion","title":"Everything at Once -- Multi-modal Fusion Transformer for Video Retrieval","date":"2021-12-08","arxiv_id":"2112.04446","repositories_listed":1,"syntology":null},{"url":"/paper/prompting-visual-language-models-for","slug":"prompting-visual-language-models-for","title":"Prompting Visual-Language Models for Efficient Video Understanding","date":"2021-12-08","arxiv_id":"2112.04478","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/prompting-visual-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2112.04478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.04478"}},"official":{"repos":["ju-chen/Efficient-Prompt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/lightweight-attentional-feature-fusion-for","slug":"lightweight-attentional-feature-fusion-for","title":"Lightweight Attentional Feature Fusion: A New Baseline for Text-to-Video Retrieval","date":"2021-12-03","arxiv_id":"2112.01832","repositories_listed":1,"syntology":null},{"url":"/paper/violet-end-to-end-video-language-transformers","slug":"violet-end-to-end-video-language-transformers","title":"VIOLET : End-to-End Video-Language Transformers with Masked Visual-token Modeling","date":"2021-11-24","arxiv_id":"2111.12681","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/violet-end-to-end-video-language-transformers#ran","syntology_url":"https://syntology.ai/paper/2111.12681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12681"}},"official":{"repos":["tsujuifu/pytorch_violet"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/advancing-high-resolution-video-language","slug":"advancing-high-resolution-video-language","title":"Advancing High-Resolution Video-Language Representation with Large-Scale Video Transcriptions","date":"2021-11-19","arxiv_id":"2111.10337","repositories_listed":1,"syntology":null},{"url":"/paper/video-and-text-matching-with-conditioned","slug":"video-and-text-matching-with-conditioned","title":"Video and Text Matching with Conditioned Embeddings","date":"2021-10-21","arxiv_id":"2110.11298","repositories_listed":1,"syntology":null},{"url":"/paper/conquer-contextual-query-aware-ranking-for","slug":"conquer-contextual-query-aware-ranking-for","title":"CONQUER: Contextual Query-aware Ranking for Video Corpus Moment Retrieval","date":"2021-09-21","arxiv_id":"2109.10016","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conquer-contextual-query-aware-ranking-for#ran","syntology_url":"https://syntology.ai/paper/2109.10016","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10016"}},"official":{"repos":["houzhijian/conquer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-contrastive-learning-with-global","slug":"video-contrastive-learning-with-global","title":"Video Contrastive Learning with Global Context","date":"2021-08-05","arxiv_id":"2108.02722","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/video-contrastive-learning-with-global#ran","syntology_url":"https://syntology.ai/paper/2108.02722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.02722"}},"official":{"repos":["amazon-research/video-contrastive-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/how-incomplete-is-contrastive-learning","slug":"how-incomplete-is-contrastive-learning","title":"Inter-intra Variant Dual Representations forSelf-supervised Video Recognition","date":"2021-07-02","arxiv_id":"2107.01194","repositories_listed":1,"syntology":null},{"url":"/paper/dns-distill-and-select-for-efficient-and","slug":"dns-distill-and-select-for-efficient-and","title":"DnS: Distill-and-Select for Efficient and Accurate Video Indexing and Retrieval","date":"2021-06-24","arxiv_id":"2106.13266","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/dns-distill-and-select-for-efficient-and#ran","syntology_url":"https://syntology.ai/paper/2106.13266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.13266"}},"official":{"repos":["mever-team/distill-and-select"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}}],"record_sha256":"a2a06790de313a9da434562421dd7008f15f4ab39cd0918bd546295ea08da050","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}