{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-retrieval/papers/ran/1","list_of":"/task/video-retrieval","task":"Video Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,95],"of":95,"counts":{"archive_papers_tagged":486,"with_a_code_link":255,"where_syntology_ran_a_sample":95,"not_listed_spam_title":0,"listed":486,"listed_where_code_ran":95,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":76,"every_run_a_failure_of_syntologys_instrument":19,"listed_with_a_run_with_no_instrument_failure":76,"listed_every_run_a_failure_of_syntologys_instrument":19,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-retrieval/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/respec-relevance-and-specificity-grounded","slug":"respec-relevance-and-specificity-grounded","title":"ReSpec: Relevance and Specificity Grounded Online Filtering for Learning on Video-Text Data Streams","date":"2025-04-21","arxiv_id":"2504.14875","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/respec-relevance-and-specificity-grounded#ran","syntology_url":"https://syntology.ai/paper/2504.14875","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.14875"}},"official":{"repos":["cdjkim/respec"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-efficient-partially-relevant-video","slug":"towards-efficient-partially-relevant-video","title":"Towards Efficient Partially Relevant Video Retrieval with Active Moment Discovering","date":"2025-04-15","arxiv_id":"2504.10920","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-efficient-partially-relevant-video#ran","syntology_url":"https://syntology.ai/paper/2504.10920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10920"}},"official":{"repos":["songpipi/amdnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tc-mgc-text-conditioned-multi-grained","slug":"tc-mgc-text-conditioned-multi-grained","title":"TC-MGC: Text-Conditioned Multi-Grained Contrastive Learning for Text-Video Retrieval","date":"2025-04-07","arxiv_id":"2504.04707","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tc-mgc-text-conditioned-multi-grained#ran","syntology_url":"https://syntology.ai/paper/2504.04707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.04707"}},"official":{"repos":["jingxiaolun/tc-mgc"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/videorope-what-makes-for-good-video-rotary","slug":"videorope-what-makes-for-good-video-rotary","title":"VideoRoPE: What Makes for Good Video Rotary Position Embedding?","date":"2025-02-07","arxiv_id":"2502.05173","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videorope-what-makes-for-good-video-rotary#ran","syntology_url":"https://syntology.ai/paper/2502.05173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05173"}},"official":{"repos":["wiselnn570/videorope"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gramian-multimodal-representation-learning","slug":"gramian-multimodal-representation-learning","title":"Gramian Multimodal Representation Learning and Alignment","date":"2024-12-16","arxiv_id":"2412.11959","repositories_listed":2,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":1,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 1 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/gramian-multimodal-representation-learning#ran","syntology_url":"https://syntology.ai/paper/2412.11959","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11959"}},"official":{"repos":["ispamm/GRAM"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/video-rag-visually-aligned-retrieval","slug":"video-rag-visually-aligned-retrieval","title":"Video-RAG: Visually-aligned Retrieval-Augmented Long Video Comprehension","date":"2024-11-20","arxiv_id":"2411.13093","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-rag-visually-aligned-retrieval#ran","syntology_url":"https://syntology.ai/paper/2411.13093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13093"}},"official":{"repos":["leon1207/video-rag-master"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/tokenbinder-text-video-retrieval-with-one-to","slug":"tokenbinder-text-video-retrieval-with-one-to","title":"TokenBinder: Text-Video Retrieval with One-to-Many Alignment Paradigm","date":"2024-09-30","arxiv_id":"2409.19865","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tokenbinder-text-video-retrieval-with-one-to#ran","syntology_url":"https://syntology.ai/paper/2409.19865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.19865"}},"official":{"repos":["bingqingzhang/TokenBinder"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tempme-video-temporal-token-merging-for","slug":"tempme-video-temporal-token-merging-for","title":"TempMe: Video Temporal Token Merging for Efficient Text-Video Retrieval","date":"2024-09-02","arxiv_id":"2409.01156","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":3,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tempme-video-temporal-token-merging-for#ran","syntology_url":"https://syntology.ai/paper/2409.01156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.01156"}},"official":null}},{"url":"/paper/seds-semantically-enhanced-dual-stream","slug":"seds-semantically-enhanced-dual-stream","title":"SEDS: Semantically Enhanced Dual-Stream Encoder for Sign Language Retrieval","date":"2024-07-23","arxiv_id":"2407.16394","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/seds-semantically-enhanced-dual-stream#ran","syntology_url":"https://syntology.ai/paper/2407.16394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16394"}},"official":{"repos":["longtaojiang/seds"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/egocvr-an-egocentric-benchmark-for-fine","slug":"egocvr-an-egocentric-benchmark-for-fine","title":"EgoCVR: An Egocentric Benchmark for Fine-Grained Composed Video Retrieval","date":"2024-07-23","arxiv_id":"2407.16658","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/egocvr-an-egocentric-benchmark-for-fine#ran","syntology_url":"https://syntology.ai/paper/2407.16658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16658"}},"official":{"repos":["explainableml/egocvr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/gmmformer-v2-an-uncertainty-aware-framework","slug":"gmmformer-v2-an-uncertainty-aware-framework","title":"GMMFormer v2: An Uncertainty-aware Framework for Partially Relevant Video Retrieval","date":"2024-05-22","arxiv_id":"2405.13824","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gmmformer-v2-an-uncertainty-aware-framework#ran","syntology_url":"https://syntology.ai/paper/2405.13824","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.13824"}},"official":{"repos":["huangmozhi9527/gmmformer_v2"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/composed-video-retrieval-via-enriched-context","slug":"composed-video-retrieval-via-enriched-context","title":"Composed Video Retrieval via Enriched Context and Discriminative Embeddings","date":"2024-03-25","arxiv_id":"2403.16997","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/composed-video-retrieval-via-enriched-context#ran","syntology_url":"https://syntology.ai/paper/2403.16997","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16997"}},"official":{"repos":["omkarthawakar/composed-video-retrieval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/egoexolearn-a-dataset-for-bridging","slug":"egoexolearn-a-dataset-for-bridging","title":"EgoExoLearn: A Dataset for Bridging Asynchronous Ego- and Exo-centric View of Procedural Activities in Real World","date":"2024-03-24","arxiv_id":"2403.16182","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/egoexolearn-a-dataset-for-bridging#ran","syntology_url":"https://syntology.ai/paper/2403.16182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16182"}},"official":{"repos":["opengvlab/egoexolearn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vid-tldr-training-free-token-merging-for","slug":"vid-tldr-training-free-token-merging-for","title":"vid-TLDR: Training Free Token merging for Light-weight Video Transformer","date":"2024-03-20","arxiv_id":"2403.13347","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vid-tldr-training-free-token-merging-for#ran","syntology_url":"https://syntology.ai/paper/2403.13347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13347"}},"official":{"repos":["mlvlab/vid-tldr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-granularity-correspondence-learning-1","slug":"multi-granularity-correspondence-learning-1","title":"Multi-granularity Correspondence Learning from Long-term Noisy Videos","date":"2024-01-30","arxiv_id":"2401.16702","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/multi-granularity-correspondence-learning-1#ran","syntology_url":"https://syntology.ai/paper/2401.16702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.16702"}},"official":null}},{"url":"/paper/dgl-dynamic-global-local-prompt-tuning-for","slug":"dgl-dynamic-global-local-prompt-tuning-for","title":"DGL: Dynamic Global-Local Prompt Tuning for Text-Video Retrieval","date":"2024-01-19","arxiv_id":"2401.10588","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/dgl-dynamic-global-local-prompt-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2401.10588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10588"}},"official":{"repos":["knightyxp/dgl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/shot2story20k-a-new-benchmark-for","slug":"shot2story20k-a-new-benchmark-for","title":"Shot2Story20K: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","date":"2023-12-16","arxiv_id":"2312.10300","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/shot2story20k-a-new-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2312.10300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10300"}},"official":{"repos":["bytedance/Shot2Story"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/let-all-be-whitened-multi-teacher","slug":"let-all-be-whitened-multi-teacher","title":"Let All be Whitened: Multi-teacher Distillation for Efficient Visual Retrieval","date":"2023-12-15","arxiv_id":"2312.09716","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/let-all-be-whitened-multi-teacher#ran","syntology_url":"https://syntology.ai/paper/2312.09716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09716"}},"official":{"repos":["maryeon/whiten_mtd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/videocon-robust-video-language-alignment-via","slug":"videocon-robust-video-language-alignment-via","title":"VideoCon: Robust Video-Language Alignment via Contrast Captions","date":"2023-11-15","arxiv_id":"2311.10111","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videocon-robust-video-language-alignment-via#ran","syntology_url":"https://syntology.ai/paper/2311.10111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10111"}},"official":{"repos":["hritikbansal/videocon"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/testa-temporal-spatial-token-aggregation-for","slug":"testa-temporal-spatial-token-aggregation-for","title":"TESTA: Temporal-Spatial Token Aggregation for Long-form Video-Language Understanding","date":"2023-10-29","arxiv_id":"2310.19060","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/testa-temporal-spatial-token-aggregation-for#ran","syntology_url":"https://syntology.ai/paper/2310.19060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19060"}},"official":{"repos":["renshuhuai-andy/testa"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/gmmformer-gaussian-mixture-model-based","slug":"gmmformer-gaussian-mixture-model-based","title":"GMMFormer: Gaussian-Mixture-Model Based Transformer for Efficient Partially Relevant Video Retrieval","date":"2023-10-08","arxiv_id":"2310.05195","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gmmformer-gaussian-mixture-model-based#ran","syntology_url":"https://syntology.ai/paper/2310.05195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05195"}},"official":{"repos":["huangmozhi9527/GMMFormer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/howtocaption-prompting-llms-to-transform","slug":"howtocaption-prompting-llms-to-transform","title":"HowToCaption: Prompting LLMs to Transform Video Annotations at Scale","date":"2023-10-07","arxiv_id":"2310.04900","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/howtocaption-prompting-llms-to-transform#ran","syntology_url":"https://syntology.ai/paper/2310.04900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04900"}},"official":{"repos":["ninatu/howtocaption"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/prototype-based-aleatoric-uncertainty-1","slug":"prototype-based-aleatoric-uncertainty-1","title":"Prototype-based Aleatoric Uncertainty Quantification for Cross-modal Retrieval","date":"2023-09-29","arxiv_id":"2309.17093","repositories_listed":1,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":10,"n_instrument":7,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":7,"n_pointer_only":8,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 1 violated, 7 with no contract checked; 7 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/prototype-based-aleatoric-uncertainty-1#ran","syntology_url":"https://syntology.ai/paper/2309.17093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17093"}},"official":{"repos":["leolee99/pau"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unified-coarse-to-fine-alignment-for-video","slug":"unified-coarse-to-fine-alignment-for-video","title":"Unified Coarse-to-Fine Alignment for Video-Text Retrieval","date":"2023-09-18","arxiv_id":"2309.10091","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unified-coarse-to-fine-alignment-for-video#ran","syntology_url":"https://syntology.ai/paper/2309.10091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.10091"}},"official":{"repos":["ziyang412/ucofia"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/covr-learning-composed-video-retrieval-from","slug":"covr-learning-composed-video-retrieval-from","title":"CoVR-2: Automatic Data Construction for Composed Video Retrieval","date":"2023-08-28","arxiv_id":"2308.14746","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/covr-learning-composed-video-retrieval-from#ran","syntology_url":"https://syntology.ai/paper/2308.14746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14746"}},"official":{"repos":["lucas-ventura/CoVR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/simple-baselines-for-interactive-video","slug":"simple-baselines-for-interactive-video","title":"Simple Baselines for Interactive Video Retrieval with Questions and Answers","date":"2023-08-21","arxiv_id":"2308.10402","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/simple-baselines-for-interactive-video#ran","syntology_url":"https://syntology.ai/paper/2308.10402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10402"}},"official":{"repos":["kevinliang888/ivr-qa-baselines"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/prompt-switch-efficient-clip-adaptation-for","slug":"prompt-switch-efficient-clip-adaptation-for","title":"Prompt Switch: Efficient CLIP Adaptation for Text-Video Retrieval","date":"2023-08-15","arxiv_id":"2308.07648","repositories_listed":1,"syntology":{"n":16,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":12,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":16,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 12 unverified","sample_list":"/paper/prompt-switch-efficient-clip-adaptation-for#ran","syntology_url":"https://syntology.ai/paper/2308.07648","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07648"}},"official":{"repos":["bladewaltz1/promptswitch"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":12,"ran_from_kinds":["official"]}}},{"url":"/paper/vast-a-vision-audio-subtitle-text-omni-1","slug":"vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","arxiv_id":"2305.18500","repositories_listed":2,"syntology":{"n":42,"n_ran":35,"n_constructed":4,"n_ran_checked":29,"n_instrument":6,"n_unverified":7,"n_honours":2,"n_violates":1,"n_no_contract":26,"n_pointer_only":8,"phrase":"35 ran (of which 4 constructed an object rather than computing a result; 29 with no instrument failure: 2 honoured, 1 violated, 26 with no contract checked; 6 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/vast-a-vision-audio-subtitle-text-omni-1#ran","syntology_url":"https://syntology.ai/paper/2305.18500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18500"}},"official":{"repos":["txh-mercury/vast"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":4,"n_ran_no_instrument_failure":12,"n_unverified":7,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/text-video-retrieval-with-disentangled","slug":"text-video-retrieval-with-disentangled","title":"Text-Video Retrieval with Disentangled Conceptualization and Set-to-Set Alignment","date":"2023-05-20","arxiv_id":"2305.12218","repositories_listed":4,"syntology":{"n":23,"n_ran":15,"n_constructed":0,"n_ran_checked":8,"n_instrument":7,"n_unverified":8,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 7 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/text-video-retrieval-with-disentangled#ran","syntology_url":"https://syntology.ai/paper/2305.12218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12218"}},"official":{"repos":["jpthu17/dicosa"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-large-cross-modal-video-retrieval-dataset","slug":"a-large-cross-modal-video-retrieval-dataset","title":"A Large Cross-Modal Video Retrieval Dataset with Reading Comprehension","date":"2023-05-05","arxiv_id":"2305.03347","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-large-cross-modal-video-retrieval-dataset#ran","syntology_url":"https://syntology.ai/paper/2305.03347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.03347"}},"official":{"repos":["callsys/textvr"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-cross-modal-knowledge-distillation-for","slug":"robust-cross-modal-knowledge-distillation-for","title":"Robust Cross-Modal Knowledge Distillation for Unconstrained Videos","date":"2023-04-16","arxiv_id":"2304.07775","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-cross-modal-knowledge-distillation-for#ran","syntology_url":"https://syntology.ai/paper/2304.07775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.07775"}},"official":{"repos":["gewu-lab/cross-modal-distillation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-video-similarity-learning","slug":"self-supervised-video-similarity-learning","title":"Self-Supervised Video Similarity Learning","date":"2023-04-06","arxiv_id":"2304.03378","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-supervised-video-similarity-learning#ran","syntology_url":"https://syntology.ai/paper/2304.03378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.03378"}},"official":{"repos":["gkordo/s2vs"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-video-moment-retrieval-and-step","slug":"hierarchical-video-moment-retrieval-and-step","title":"Hierarchical Video-Moment Retrieval and Step-Captioning","date":"2023-03-29","arxiv_id":"2303.16406","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hierarchical-video-moment-retrieval-and-step#ran","syntology_url":"https://syntology.ai/paper/2303.16406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16406"}},"official":{"repos":["j-min/HiREST"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unmasked-teacher-towards-training-efficient","slug":"unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","arxiv_id":"2303.16058","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unmasked-teacher-towards-training-efficient#ran","syntology_url":"https://syntology.ai/paper/2303.16058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16058"}},"official":{"repos":["opengvlab/unmasked_teacher"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-text-as-game-players-hierarchical","slug":"video-text-as-game-players-hierarchical","title":"Video-Text as Game Players: Hierarchical Banzhaf Interaction for Cross-Modal Representation Learning","date":"2023-03-25","arxiv_id":"2303.14369","repositories_listed":4,"syntology":{"n":16,"n_ran":12,"n_constructed":7,"n_ran_checked":11,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 7 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/video-text-as-game-players-hierarchical#ran","syntology_url":"https://syntology.ai/paper/2303.14369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14369"}},"official":{"repos":["jpthu17/HBI"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/aligning-step-by-step-instructional-diagrams","slug":"aligning-step-by-step-instructional-diagrams","title":"Aligning Step-by-Step Instructional Diagrams to Video Demonstrations","date":"2023-03-24","arxiv_id":"2303.13800","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aligning-step-by-step-instructional-diagrams#ran","syntology_url":"https://syntology.ai/paper/2303.13800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13800"}},"official":{"repos":["DavidZhang73/AssemblyVideoManualAlignment"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meltr-meta-loss-transformer-for-learning-to","slug":"meltr-meta-loss-transformer-for-learning-to","title":"MELTR: Meta Loss Transformer for Learning to Fine-tune Video Foundation Models","date":"2023-03-23","arxiv_id":"2303.13009","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/meltr-meta-loss-transformer-for-learning-to#ran","syntology_url":"https://syntology.ai/paper/2303.13009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13009"}},"official":{"repos":["mlvlab/MELTR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusionret-generative-text-video-retrieval","slug":"diffusionret-generative-text-video-retrieval","title":"DiffusionRet: Generative Text-Video Retrieval with Diffusion Model","date":"2023-03-17","arxiv_id":"2303.09867","repositories_listed":4,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diffusionret-generative-text-video-retrieval#ran","syntology_url":"https://syntology.ai/paper/2303.09867","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.09867"}},"official":{"repos":["jpthu17/diffusionret"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mplug-2-a-modularized-multi-modal-foundation","slug":"mplug-2-a-modularized-multi-modal-foundation","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","date":"2023-02-01","arxiv_id":"2302.00402","repositories_listed":4,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":9,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mplug-2-a-modularized-multi-modal-foundation#ran","syntology_url":"https://syntology.ai/paper/2302.00402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00402"}},"official":{"repos":["alibaba/AliceMind"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/zorro-the-masked-multimodal-transformer","slug":"zorro-the-masked-multimodal-transformer","title":"Zorro: the masked multimodal transformer","date":"2023-01-23","arxiv_id":"2301.09595","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zorro-the-masked-multimodal-transformer#ran","syntology_url":"https://syntology.ai/paper/2301.09595","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.09595"}},"official":null}},{"url":"/paper/cap4video-what-can-auxiliary-captions-do-for","slug":"cap4video-what-can-auxiliary-captions-do-for","title":"Cap4Video: What Can Auxiliary Captions Do for Text-Video Retrieval?","date":"2022-12-31","arxiv_id":"2301.00184","repositories_listed":4,"syntology":{"n":25,"n_ran":20,"n_constructed":0,"n_ran_checked":12,"n_instrument":8,"n_unverified":5,"n_honours":2,"n_violates":1,"n_no_contract":9,"n_pointer_only":11,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 2 honoured, 1 violated, 9 with no contract checked; 8 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/cap4video-what-can-auxiliary-captions-do-for#ran","syntology_url":"https://syntology.ai/paper/2301.00184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.00184"}},"official":{"repos":["whwu95/Cap4Video"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/tempclr-temporal-alignment-representation","slug":"tempclr-temporal-alignment-representation","title":"TempCLR: Temporal Alignment Representation with Contrastive Learning","date":"2022-12-28","arxiv_id":"2212.13738","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tempclr-temporal-alignment-representation#ran","syntology_url":"https://syntology.ai/paper/2212.13738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.13738"}},"official":{"repos":["yyuncong/tempclr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/internvideo-general-video-foundation-models","slug":"internvideo-general-video-foundation-models","title":"InternVideo: General Video Foundation Models via Generative and Discriminative Learning","date":"2022-12-06","arxiv_id":"2212.03191","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internvideo-general-video-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2212.03191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.03191"}},"official":{"repos":["opengvlab/internvideo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","slug":"x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","arxiv_id":"2211.12402","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/x-2-vlm-all-in-one-pre-trained-model-for#ran","syntology_url":"https://syntology.ai/paper/2211.12402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12402"}},"official":{"repos":["zengyan-97/x2-vlm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/expectation-maximization-contrastive-learning","slug":"expectation-maximization-contrastive-learning","title":"Expectation-Maximization Contrastive Learning for Compact Video-and-Language Representations","date":"2022-11-21","arxiv_id":"2211.11427","repositories_listed":4,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/expectation-maximization-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2211.11427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11427"}},"official":{"repos":["jpthu17/emcl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/cross-modal-adapter-for-text-video-retrieval","slug":"cross-modal-adapter-for-text-video-retrieval","title":"Cross-Modal Adapter for Text-Video Retrieval","date":"2022-11-17","arxiv_id":"2211.09623","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":10,"n_pointer_only":6,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 2 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-modal-adapter-for-text-video-retrieval#ran","syntology_url":"https://syntology.ai/paper/2211.09623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09623"}},"official":{"repos":["leaplabthu/cross-modal-adapter"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/long-form-video-language-pre-training-with","slug":"long-form-video-language-pre-training-with","title":"Long-Form Video-Language Pre-Training with Multimodal Temporal Contrastive Learning","date":"2022-10-12","arxiv_id":"2210.06031","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/long-form-video-language-pre-training-with#ran","syntology_url":"https://syntology.ai/paper/2210.06031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06031"}},"official":{"repos":["microsoft/xpretrain"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tvlt-textless-vision-language-transformer","slug":"tvlt-textless-vision-language-transformer","title":"TVLT: Textless Vision-Language Transformer","date":"2022-09-28","arxiv_id":"2209.14156","repositories_listed":3,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tvlt-textless-vision-language-transformer#ran","syntology_url":"https://syntology.ai/paper/2209.14156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.14156"}},"official":{"repos":["zinengtang/tvlt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/clip-vip-adapting-pre-trained-image-text","slug":"clip-vip-adapting-pre-trained-image-text","title":"CLIP-ViP: Adapting Pre-trained Image-Text Model to Video-Language Representation Alignment","date":"2022-09-14","arxiv_id":"2209.06430","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clip-vip-adapting-pre-trained-image-text#ran","syntology_url":"https://syntology.ai/paper/2209.06430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.06430"}},"official":{"repos":["microsoft/xpretrain"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ts2-net-token-shift-and-selection-transformer","slug":"ts2-net-token-shift-and-selection-transformer","title":"TS2-Net: Token Shift and Selection Transformer for Text-Video Retrieval","date":"2022-07-16","arxiv_id":"2207.07852","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/ts2-net-token-shift-and-selection-transformer#ran","syntology_url":"https://syntology.ai/paper/2207.07852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.07852"}},"official":{"repos":["yuqi657/ts2_net"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/x-clip-end-to-end-multi-grained-contrastive","slug":"x-clip-end-to-end-multi-grained-contrastive","title":"X-CLIP: End-to-End Multi-grained Contrastive Learning for Video-Text Retrieval","date":"2022-07-15","arxiv_id":"2207.07285","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/x-clip-end-to-end-multi-grained-contrastive#ran","syntology_url":"https://syntology.ai/paper/2207.07285","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.07285"}},"official":{"repos":["xuguohai/X-CLIP"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/revealing-single-frame-bias-for-video-and","slug":"revealing-single-frame-bias-for-video-and","title":"Revealing Single Frame Bias for Video-and-Language Learning","date":"2022-06-07","arxiv_id":"2206.03428","repositories_listed":2,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revealing-single-frame-bias-for-video-and#ran","syntology_url":"https://syntology.ai/paper/2206.03428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.03428"}},"official":{"repos":["jayleicn/singularity"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/revisiting-the-video-in-video-language","slug":"revisiting-the-video-in-video-language","title":"Revisiting the \"Video\" in Video-Language Understanding","date":"2022-06-03","arxiv_id":"2206.01720","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/revisiting-the-video-in-video-language#ran","syntology_url":"https://syntology.ai/paper/2206.01720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01720"}},"official":null}},{"url":"/paper/cross-architecture-self-supervised-video","slug":"cross-architecture-self-supervised-video","title":"Cross-Architecture Self-supervised Video Representation Learning","date":"2022-05-26","arxiv_id":"2205.13313","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/cross-architecture-self-supervised-video#ran","syntology_url":"https://syntology.ai/paper/2205.13313","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13313"}},"official":{"repos":["guoshengcv/cacl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/coca-contrastive-captioners-are-image-text","slug":"coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","arxiv_id":"2205.01917","repositories_listed":6,"syntology":{"n":17,"n_ran":10,"n_constructed":5,"n_ran_checked":10,"n_instrument":0,"n_unverified":7,"n_honours":2,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/coca-contrastive-captioners-are-image-text#ran","syntology_url":"https://syntology.ai/paper/2205.01917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01917"}},"official":null}},{"url":"/paper/relevance-based-margin-for-contrastively","slug":"relevance-based-margin-for-contrastively","title":"Relevance-based Margin for Contrastively-trained Video Retrieval Models","date":"2022-04-27","arxiv_id":"2204.13001","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/relevance-based-margin-for-contrastively#ran","syntology_url":"https://syntology.ai/paper/2204.13001","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.13001"}},"official":{"repos":["aranciokov/relevancemargin-icmr22"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/eclipse-efficient-long-range-video-retrieval","slug":"eclipse-efficient-long-range-video-retrieval","title":"ECLIPSE: Efficient Long-range Video Retrieval using Sight and Sound","date":"2022-04-06","arxiv_id":"2204.02874","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/eclipse-efficient-long-range-video-retrieval#ran","syntology_url":"https://syntology.ai/paper/2204.02874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02874"}},"official":{"repos":["GenjiB/ECLIPSE"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/temporal-alignment-networks-for-long-term","slug":"temporal-alignment-networks-for-long-term","title":"Temporal Alignment Networks for Long-term Video","date":"2022-04-06","arxiv_id":"2204.02968","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/temporal-alignment-networks-for-long-term#ran","syntology_url":"https://syntology.ai/paper/2204.02968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02968"}},"official":null}},{"url":"/paper/x-pool-cross-modal-language-video-attention","slug":"x-pool-cross-modal-language-video-attention","title":"X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval","date":"2022-03-28","arxiv_id":"2203.15086","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/x-pool-cross-modal-language-video-attention#ran","syntology_url":"https://syntology.ai/paper/2203.15086","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15086"}},"official":{"repos":["layer6ai-labs/xpool"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fitclip-refining-large-scale-pretrained-image","slug":"fitclip-refining-large-scale-pretrained-image","title":"FitCLIP: Refining Large-Scale Pretrained Image-Text Models for Zero-Shot Video Understanding Tasks","date":"2022-03-24","arxiv_id":"2203.13371","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fitclip-refining-large-scale-pretrained-image#ran","syntology_url":"https://syntology.ai/paper/2203.13371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13371"}},"official":{"repos":["bryant1410/fitclip","bryant1410/tclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/disentangled-representation-learning-for-text","slug":"disentangled-representation-learning-for-text","title":"Disentangled Representation Learning for Text-Video Retrieval","date":"2022-03-14","arxiv_id":"2203.07111","repositories_listed":2,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":3,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/disentangled-representation-learning-for-text#ran","syntology_url":"https://syntology.ai/paper/2203.07111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.07111"}},"official":{"repos":["foolwood/DRL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/show-me-more-details-discovering-hierarchies","slug":"show-me-more-details-discovering-hierarchies","title":"Show Me More Details: Discovering Hierarchies of Procedures from Semi-structured Web Data","date":"2022-03-14","arxiv_id":"2203.07264","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/show-me-more-details-discovering-hierarchies#ran","syntology_url":"https://syntology.ai/paper/2203.07264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.07264"}},"official":{"repos":["shuyanzhou/wikihow_hierarchy"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/bridgeformer-bridging-video-text-retrieval","slug":"bridgeformer-bridging-video-text-retrieval","title":"Bridging Video-text Retrieval with Multiple Choice Questions","date":"2022-01-13","arxiv_id":"2201.04850","repositories_listed":2,"syntology":{"n":24,"n_ran":13,"n_constructed":8,"n_ran_checked":9,"n_instrument":4,"n_unverified":11,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":6,"phrase":"13 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/bridgeformer-bridging-video-text-retrieval#ran","syntology_url":"https://syntology.ai/paper/2201.04850","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.04850"}},"official":{"repos":["tencentarc/mcq"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/cross-modal-retrieval-with-querybank","slug":"cross-modal-retrieval-with-querybank","title":"Cross Modal Retrieval with Querybank Normalisation","date":"2021-12-23","arxiv_id":"2112.12777","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cross-modal-retrieval-with-querybank#ran","syntology_url":"https://syntology.ai/paper/2112.12777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.12777"}},"official":{"repos":["ioanacroi/qb-norm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/prompting-visual-language-models-for","slug":"prompting-visual-language-models-for","title":"Prompting Visual-Language Models for Efficient Video Understanding","date":"2021-12-08","arxiv_id":"2112.04478","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/prompting-visual-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2112.04478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.04478"}},"official":{"repos":["ju-chen/Efficient-Prompt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/conquer-contextual-query-aware-ranking-for","slug":"conquer-contextual-query-aware-ranking-for","title":"CONQUER: Contextual Query-aware Ranking for Video Corpus Moment Retrieval","date":"2021-09-21","arxiv_id":"2109.10016","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conquer-contextual-query-aware-ranking-for#ran","syntology_url":"https://syntology.ai/paper/2109.10016","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10016"}},"official":{"repos":["houzhijian/conquer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-contrastive-learning-with-global","slug":"video-contrastive-learning-with-global","title":"Video Contrastive Learning with Global Context","date":"2021-08-05","arxiv_id":"2108.02722","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/video-contrastive-learning-with-global#ran","syntology_url":"https://syntology.ai/paper/2108.02722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.02722"}},"official":{"repos":["amazon-research/video-contrastive-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/value-a-multi-task-benchmark-for-video-and","slug":"value-a-multi-task-benchmark-for-video-and","title":"VALUE: A Multi-Task Benchmark for Video-and-Language Understanding Evaluation","date":"2021-06-08","arxiv_id":"2106.04632","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/value-a-multi-task-benchmark-for-video-and#ran","syntology_url":"https://syntology.ai/paper/2106.04632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04632"}},"official":{"repos":["VALUE-Leaderboard/StarterCode"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-clustering-networks-for-self","slug":"multimodal-clustering-networks-for-self","title":"Multimodal Clustering Networks for Self-supervised Learning from Unlabeled Videos","date":"2021-04-26","arxiv_id":"2104.12671","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-clustering-networks-for-self#ran","syntology_url":"https://syntology.ai/paper/2104.12671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.12671"}},"official":{"repos":["brian7685/Multimodal-Clustering-Network"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vatt-transformers-for-multimodal-self","slug":"vatt-transformers-for-multimodal-self","title":"VATT: Transformers for Multimodal Self-Supervised Learning from Raw Video, Audio and Text","date":"2021-04-22","arxiv_id":"2104.11178","repositories_listed":5,"syntology":{"n":8,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vatt-transformers-for-multimodal-self#ran","syntology_url":"https://syntology.ai/paper/2104.11178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.11178"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/clip4clip-an-empirical-study-of-clip-for-end","slug":"clip4clip-an-empirical-study-of-clip-for-end","title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval","date":"2021-04-18","arxiv_id":"2104.08860","repositories_listed":5,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/clip4clip-an-empirical-study-of-clip-for-end#ran","syntology_url":"https://syntology.ai/paper/2104.08860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08860"}},"official":{"repos":["ArrowLuo/CLIP4Clip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/teachtext-crossmodal-generalized-distillation","slug":"teachtext-crossmodal-generalized-distillation","title":"TEACHTEXT: CrossModal Generalized Distillation for Text-Video Retrieval","date":"2021-04-16","arxiv_id":"2104.08271","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/teachtext-crossmodal-generalized-distillation#ran","syntology_url":"https://syntology.ai/paper/2104.08271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08271"}},"official":{"repos":["albanie/collaborative-experts"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/frozen-in-time-a-joint-video-and-image","slug":"frozen-in-time-a-joint-video-and-image","title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval","date":"2021-04-01","arxiv_id":"2104.00650","repositories_listed":5,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/frozen-in-time-a-joint-video-and-image#ran","syntology_url":"https://syntology.ai/paper/2104.00650","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.00650"}},"official":{"repos":["m-bain/frozen-in-time"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mdmmt-multidomain-multimodal-transformer-for","slug":"mdmmt-multidomain-multimodal-transformer-for","title":"MDMMT: Multidomain Multimodal Transformer for Video Retrieval","date":"2021-03-19","arxiv_id":"2103.10699","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mdmmt-multidomain-multimodal-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2103.10699","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.10699"}},"official":{"repos":["papermsucode/mdmmt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-semantic-similarity-in-video-retrieval","slug":"on-semantic-similarity-in-video-retrieval","title":"On Semantic Similarity in Video Retrieval","date":"2021-03-18","arxiv_id":"2103.10095","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-semantic-similarity-in-video-retrieval#ran","syntology_url":"https://syntology.ai/paper/2103.10095","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.10095"}},"official":{"repos":["mwray/Semantic-Video-Retrieval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/coot-cooperative-hierarchical-transformer-for","slug":"coot-cooperative-hierarchical-transformer-for","title":"COOT: Cooperative Hierarchical Transformer for Video-Text Representation Learning","date":"2020-11-01","arxiv_id":"2011.00597","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coot-cooperative-hierarchical-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2011.00597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00597"}},"official":{"repos":["gingsi/coot-videotext"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-co-training-for-video","slug":"self-supervised-co-training-for-video","title":"Self-supervised Co-training for Video Representation Learning","date":"2020-10-19","arxiv_id":"2010.09709","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-co-training-for-video#ran","syntology_url":"https://syntology.ai/paper/2010.09709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09709"}},"official":{"repos":["TengdaHan/CoCLR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-video-representation-learning-5","slug":"self-supervised-video-representation-learning-5","title":"Self-supervised Video Representation Learning by Uncovering Spatio-temporal Statistics","date":"2020-08-31","arxiv_id":"2008.13426","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-video-representation-learning-5#ran","syntology_url":"https://syntology.ai/paper/2008.13426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.13426"}},"official":{"repos":["laura-wang/video_repres_sts"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/self-supervised-video-representation-learning-3","slug":"self-supervised-video-representation-learning-3","title":"Self-supervised Video Representation Learning Using Inter-intra Contrastive Framework","date":"2020-08-06","arxiv_id":"2008.02531","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-video-representation-learning-3#ran","syntology_url":"https://syntology.ai/paper/2008.02531","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.02531"}},"official":{"repos":["BestJuly/Inter-intra-video-contrastive-learning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/memory-augmented-dense-predictive-coding-for","slug":"memory-augmented-dense-predictive-coding-for","title":"Memory-augmented Dense Predictive Coding for Video Representation Learning","date":"2020-08-03","arxiv_id":"2008.01065","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/memory-augmented-dense-predictive-coding-for#ran","syntology_url":"https://syntology.ai/paper/2008.01065","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.01065"}},"official":{"repos":["TengdaHan/MemDPC"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-transformer-for-video-retrieval","slug":"multi-modal-transformer-for-video-retrieval","title":"Multi-modal Transformer for Video Retrieval","date":"2020-07-21","arxiv_id":"2007.10639","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-modal-transformer-for-video-retrieval#ran","syntology_url":"https://syntology.ai/paper/2007.10639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.10639"}},"official":null}},{"url":"/paper/video-playback-rate-perception-for-self-1","slug":"video-playback-rate-perception-for-self-1","title":"Video Playback Rate Perception for Self-supervisedSpatio-Temporal Representation Learning","date":"2020-06-20","arxiv_id":"2006.11476","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/video-playback-rate-perception-for-self-1#ran","syntology_url":"https://syntology.ai/paper/2006.11476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11476"}},"official":{"repos":["yuanyao366/PRP"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/condensed-movies-story-based-retrieval-with","slug":"condensed-movies-story-based-retrieval-with","title":"Condensed Movies: Story Based Retrieval with Contextual Embeddings","date":"2020-05-08","arxiv_id":"2005.04208","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/condensed-movies-story-based-retrieval-with#ran","syntology_url":"https://syntology.ai/paper/2005.04208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.04208"}},"official":null}},{"url":"/paper/hero-hierarchical-encoder-for-video-language","slug":"hero-hierarchical-encoder-for-video-language","title":"HERO: Hierarchical Encoder for Video+Language Omni-representation Pre-training","date":"2020-05-01","arxiv_id":"2005.00200","repositories_listed":3,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hero-hierarchical-encoder-for-video-language#ran","syntology_url":"https://syntology.ai/paper/2005.00200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.00200"}},"official":{"repos":["linjieli222/HERO"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/univilm-a-unified-video-and-language-pre","slug":"univilm-a-unified-video-and-language-pre","title":"UniVL: A Unified Video and Language Pre-Training Model for Multimodal Understanding and Generation","date":"2020-02-15","arxiv_id":"2002.06353","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/univilm-a-unified-video-and-language-pre#ran","syntology_url":"https://syntology.ai/paper/2002.06353","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06353"}},"official":{"repos":["microsoft/UniVL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/tvr-a-large-scale-dataset-for-video-subtitle","slug":"tvr-a-large-scale-dataset-for-video-subtitle","title":"TVR: A Large-Scale Dataset for Video-Subtitle Moment Retrieval","date":"2020-01-24","arxiv_id":"2001.09099","repositories_listed":2,"syntology":{"n":19,"n_ran":19,"n_constructed":0,"n_ran_checked":16,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":14,"n_pointer_only":3,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 2 honoured, 0 violated, 14 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tvr-a-large-scale-dataset-for-video-subtitle#ran","syntology_url":"https://syntology.ai/paper/2001.09099","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.09099"}},"official":{"repos":["jayleicn/TVCaption","jayleicn/TVRetrieval"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-cloze-procedure-for-self-supervised","slug":"video-cloze-procedure-for-self-supervised","title":"Video Cloze Procedure for Self-Supervised Spatio-Temporal Learning","date":"2020-01-02","arxiv_id":"2001.00294","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-cloze-procedure-for-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2001.00294","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00294"}},"official":null}},{"url":"/paper/end-to-end-learning-of-visual-representations","slug":"end-to-end-learning-of-visual-representations","title":"End-to-End Learning of Visual Representations from Uncurated Instructional Videos","date":"2019-12-13","arxiv_id":"1912.06430","repositories_listed":4,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/end-to-end-learning-of-visual-representations#ran","syntology_url":"https://syntology.ai/paper/1912.06430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.06430"}},"official":{"repos":["antoine77340/MIL-NCE_HowTo100M"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/visil-fine-grained-spatio-temporal-video","slug":"visil-fine-grained-spatio-temporal-video","title":"ViSiL: Fine-grained Spatio-Temporal Video Similarity Learning","date":"2019-08-20","arxiv_id":"1908.07410","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/visil-fine-grained-spatio-temporal-video#ran","syntology_url":"https://syntology.ai/paper/1908.07410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.07410"}},"official":{"repos":["MKLab-ITI/visil"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/central-similarity-hashing-via-hadamard","slug":"central-similarity-hashing-via-hadamard","title":"Central Similarity Quantization for Efficient Image and Video Retrieval","date":"2019-08-01","arxiv_id":"1908.00347","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/central-similarity-hashing-via-hadamard#ran","syntology_url":"https://syntology.ai/paper/1908.00347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.00347"}},"official":{"repos":["yuanli2333/Hadamard-Matrix-for-hashing"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/use-what-you-have-video-retrieval-using","slug":"use-what-you-have-video-retrieval-using","title":"Use What You Have: Video Retrieval Using Representations From Collaborative Experts","date":"2019-07-31","arxiv_id":"1907.13487","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/use-what-you-have-video-retrieval-using#ran","syntology_url":"https://syntology.ai/paper/1907.13487","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.13487"}},"official":null}},{"url":"/paper/howto100m-learning-a-text-video-embedding-by","slug":"howto100m-learning-a-text-video-embedding-by","title":"HowTo100M: Learning a Text-Video Embedding by Watching Hundred Million Narrated Video Clips","date":"2019-06-07","arxiv_id":"1906.03327","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/howto100m-learning-a-text-video-embedding-by#ran","syntology_url":"https://syntology.ai/paper/1906.03327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.03327"}},"official":null}},{"url":"/paper/dual-dense-encoding-for-zero-example-video","slug":"dual-dense-encoding-for-zero-example-video","title":"Dual Encoding for Zero-Example Video Retrieval","date":"2018-09-17","arxiv_id":"1809.06181","repositories_listed":1,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":2,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dual-dense-encoding-for-zero-example-video#ran","syntology_url":"https://syntology.ai/paper/1809.06181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.06181"}},"official":{"repos":["danieljf24/dual_encoding"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-joint-sequence-fusion-model-for-video","slug":"a-joint-sequence-fusion-model-for-video","title":"A Joint Sequence Fusion Model for Video Question Answering and Retrieval","date":"2018-08-07","arxiv_id":"1808.02559","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-joint-sequence-fusion-model-for-video#ran","syntology_url":"https://syntology.ai/paper/1808.02559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.02559"}},"official":null}},{"url":"/paper/talking-face-generation-by-adversarially","slug":"talking-face-generation-by-adversarially","title":"Talking Face Generation by Adversarially Disentangled Audio-Visual Representation","date":"2018-07-20","arxiv_id":"1807.07860","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/talking-face-generation-by-adversarially#ran","syntology_url":"https://syntology.ai/paper/1807.07860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.07860"}},"official":null}}],"record_sha256":"b6b8b987cab14bc2c670c2226c8d482f25a01b62f672d129d8dd34cbee8b0059","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}