{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-retrieval/papers/3","list_of":"/task/video-retrieval","task":"Video Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":5,"rows_per_page":100,"rows":[201,300],"of":486,"counts":{"archive_papers_tagged":486,"with_a_code_link":255,"where_syntology_ran_a_sample":95,"not_listed_spam_title":0,"listed":486,"listed_where_code_ran":95,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":76,"every_run_a_failure_of_syntologys_instrument":19,"listed_with_a_run_with_no_instrument_failure":76,"listed_every_run_a_failure_of_syntologys_instrument":19,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-retrieval","prev":"/task/video-retrieval/papers/2","next":"/task/video-retrieval/papers/4","papers":[{"url":"/paper/clip2video-mastering-video-text-retrieval-via","slug":"clip2video-mastering-video-text-retrieval-via","title":"CLIP2Video: Mastering Video-Text Retrieval via Image CLIP","date":"2021-06-21","arxiv_id":"2106.11097","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/clip2video-mastering-video-text-retrieval-via#ran","syntology_url":"https://syntology.ai/paper/2106.11097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.11097"}},"official":{"repos":["CryhanFang/CLIP2Video"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/self-supervised-video-hashing-via","slug":"self-supervised-video-hashing-via","title":"Self-Supervised Video Hashing via Bidirectional Transformers","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-video-representation-learning-7","slug":"self-supervised-video-representation-learning-7","title":"Self-supervised Video Representation Learning with Cross-Stream Prototypical Contrasting","date":"2021-06-18","arxiv_id":"2106.10137","repositories_listed":1,"syntology":null},{"url":"/paper/value-a-multi-task-benchmark-for-video-and","slug":"value-a-multi-task-benchmark-for-video-and","title":"VALUE: A Multi-Task Benchmark for Video-and-Language Understanding Evaluation","date":"2021-06-08","arxiv_id":"2106.04632","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/value-a-multi-task-benchmark-for-video-and#ran","syntology_url":"https://syntology.ai/paper/2106.04632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04632"}},"official":{"repos":["VALUE-Leaderboard/StarterCode"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/decembert-learning-from-noisy-instructional","slug":"decembert-learning-from-noisy-instructional","title":"DeCEMBERT: Learning from Noisy Instructional Videos via Dense Captions and Entropy Minimization","date":"2021-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/vlm-task-agnostic-video-language-model-pre","slug":"vlm-task-agnostic-video-language-model-pre","title":"VLM: Task-agnostic Video-Language Model Pre-training for Video Understanding","date":"2021-05-20","arxiv_id":"2105.09996","repositories_listed":1,"syntology":null},{"url":"/paper/trecvid-2020-a-comprehensive-campaign-for","slug":"trecvid-2020-a-comprehensive-campaign-for","title":"TRECVID 2020: A comprehensive campaign for evaluating video retrieval tasks across multiple application domains","date":"2021-04-27","arxiv_id":"2104.13473","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-clustering-networks-for-self","slug":"multimodal-clustering-networks-for-self","title":"Multimodal Clustering Networks for Self-supervised Learning from Unlabeled Videos","date":"2021-04-26","arxiv_id":"2104.12671","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-clustering-networks-for-self#ran","syntology_url":"https://syntology.ai/paper/2104.12671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.12671"}},"official":{"repos":["brian7685/Multimodal-Clustering-Network"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/t2vlad-global-local-sequence-alignment-for","slug":"t2vlad-global-local-sequence-alignment-for","title":"T2VLAD: Global-Local Sequence Alignment for Text-Video Retrieval","date":"2021-04-20","arxiv_id":"2104.10054","repositories_listed":1,"syntology":null},{"url":"/paper/teachtext-crossmodal-generalized-distillation","slug":"teachtext-crossmodal-generalized-distillation","title":"TEACHTEXT: CrossModal Generalized Distillation for Text-Video Retrieval","date":"2021-04-16","arxiv_id":"2104.08271","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/teachtext-crossmodal-generalized-distillation#ran","syntology_url":"https://syntology.ai/paper/2104.08271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08271"}},"official":{"repos":["albanie/collaborative-experts"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/object-priors-for-classifying-and-localizing","slug":"object-priors-for-classifying-and-localizing","title":"Object Priors for Classifying and Localizing Unseen Actions","date":"2021-04-10","arxiv_id":"2104.04715","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-video-representation-learning-6","slug":"self-supervised-video-representation-learning-6","title":"Self-supervised Video Representation Learning by Context and Motion Decoupling","date":"2021-04-02","arxiv_id":"2104.00862","repositories_listed":1,"syntology":null},{"url":"/paper/rudder-a-cross-lingual-video-and-text","slug":"rudder-a-cross-lingual-video-and-text","title":"Rudder: A Cross Lingual Video and Text Retrieval Dataset","date":"2021-03-09","arxiv_id":"2103.05457","repositories_listed":1,"syntology":null},{"url":"/paper/a-straightforward-framework-for-video","slug":"a-straightforward-framework-for-video","title":"A Straightforward Framework For Video Retrieval Using CLIP","date":"2021-02-24","arxiv_id":"2102.12443","repositories_listed":1,"syntology":null},{"url":"/paper/seqnet-learning-descriptors-for-sequence","slug":"seqnet-learning-descriptors-for-sequence","title":"SeqNet: Learning Descriptors for Sequence-based Hierarchical Place Recognition","date":"2021-02-23","arxiv_id":"2102.11603","repositories_listed":1,"syntology":null},{"url":"/paper/win-fail-action-recognition","slug":"win-fail-action-recognition","title":"Win-Fail Action Recognition","date":"2021-02-15","arxiv_id":"2102.07355","repositories_listed":1,"syntology":null},{"url":"/paper/less-is-more-clipbert-for-video-and-language","slug":"less-is-more-clipbert-for-video-and-language","title":"Less is More: ClipBERT for Video-and-Language Learning via Sparse Sampling","date":"2021-02-11","arxiv_id":"2102.06183","repositories_listed":1,"syntology":null},{"url":"/paper/sea-sentence-encoder-assembly-for-video","slug":"sea-sentence-encoder-assembly-for-video","title":"SEA: Sentence Encoder Assembly for Video Retrieval by Textual Queries","date":"2020-11-24","arxiv_id":"2011.12091","repositories_listed":1,"syntology":null},{"url":"/paper/graph-based-temporal-aggregation-for-video","slug":"graph-based-temporal-aggregation-for-video","title":"Graph Based Temporal Aggregation for Video Retrieval","date":"2020-11-04","arxiv_id":"2011.02426","repositories_listed":1,"syntology":null},{"url":"/paper/coot-cooperative-hierarchical-transformer-for","slug":"coot-cooperative-hierarchical-transformer-for","title":"COOT: Cooperative Hierarchical Transformer for Video-Text Representation Learning","date":"2020-11-01","arxiv_id":"2011.00597","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coot-cooperative-hierarchical-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2011.00597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00597"}},"official":{"repos":["gingsi/coot-videotext"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-video-representation-using","slug":"self-supervised-video-representation-using","title":"Pretext-Contrastive Learning: Toward Good Practices in Self-supervised Video Representation Leaning","date":"2020-10-29","arxiv_id":"2010.15464","repositories_listed":1,"syntology":null},{"url":"/paper/rspnet-relative-speed-perception-for","slug":"rspnet-relative-speed-perception-for","title":"RSPNet: Relative Speed Perception for Unsupervised Video Representation Learning","date":"2020-10-27","arxiv_id":"2011.07949","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-co-training-for-video","slug":"self-supervised-co-training-for-video","title":"Self-supervised Co-training for Video Representation Learning","date":"2020-10-19","arxiv_id":"2010.09709","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-supervised-co-training-for-video#ran","syntology_url":"https://syntology.ai/paper/2010.09709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09709"}},"official":{"repos":["TengdaHan/CoCLR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/audio-based-near-duplicate-video-retrieval","slug":"audio-based-near-duplicate-video-retrieval","title":"Audio-based Near-Duplicate Video Retrieval with Audio Similarity Learning","date":"2020-10-17","arxiv_id":"2010.08737","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-space-learning-for-language-based","slug":"hybrid-space-learning-for-language-based","title":"Dual Encoding for Video Retrieval by Text","date":"2020-09-10","arxiv_id":"2009.05381","repositories_listed":1,"syntology":null},{"url":"/paper/discriminative-residual-analysis-for-image","slug":"discriminative-residual-analysis-for-image","title":"Discriminative Residual Analysis for Image Set Classification with Posture and Age Variations","date":"2020-08-23","arxiv_id":"2008.09994","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-video-representation-learning-4","slug":"self-supervised-video-representation-learning-4","title":"Self-supervised Video Representation Learning by Pace Prediction","date":"2020-08-13","arxiv_id":"2008.05861","repositories_listed":1,"syntology":null},{"url":"/paper/context-encoding-for-video-retrieval-with","slug":"context-encoding-for-video-retrieval-with","title":"Temporal Context Aggregation for Video Retrieval with Contrastive Learning","date":"2020-08-04","arxiv_id":"2008.01334","repositories_listed":1,"syntology":null},{"url":"/paper/memory-augmented-dense-predictive-coding-for","slug":"memory-augmented-dense-predictive-coding-for","title":"Memory-augmented Dense Predictive Coding for Video Representation Learning","date":"2020-08-03","arxiv_id":"2008.01065","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/memory-augmented-dense-predictive-coding-for#ran","syntology_url":"https://syntology.ai/paper/2008.01065","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.01065"}},"official":{"repos":["TengdaHan/MemDPC"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/the-end-of-end-to-end-a-video-understanding","slug":"the-end-of-end-to-end-a-video-understanding","title":"The End-of-End-to-End: A Video Understanding Pentathlon Challenge (2020)","date":"2020-08-03","arxiv_id":"2008.00744","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-transformer-for-video-retrieval","slug":"multi-modal-transformer-for-video-retrieval","title":"Multi-modal Transformer for Video Retrieval","date":"2020-07-21","arxiv_id":"2007.10639","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-modal-transformer-for-video-retrieval#ran","syntology_url":"https://syntology.ai/paper/2007.10639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.10639"}},"official":null}},{"url":"/paper/generalized-many-way-few-shot-video","slug":"generalized-many-way-few-shot-video","title":"Generalized Few-Shot Video Classification with Video Retrieval and Feature Generation","date":"2020-07-09","arxiv_id":"2007.04755","repositories_listed":1,"syntology":null},{"url":"/paper/video-playback-rate-perception-for-self-1","slug":"video-playback-rate-perception-for-self-1","title":"Video Playback Rate Perception for Self-supervisedSpatio-Temporal Representation Learning","date":"2020-06-20","arxiv_id":"2006.11476","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/video-playback-rate-perception-for-self-1#ran","syntology_url":"https://syntology.ai/paper/2006.11476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11476"}},"official":{"repos":["yuanyao366/PRP"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/avlnet-learning-audio-visual-language","slug":"avlnet-learning-audio-visual-language","title":"AVLnet: Learning Audio-Visual Language Representations from Instructional Videos","date":"2020-06-16","arxiv_id":"2006.09199","repositories_listed":1,"syntology":null},{"url":"/paper/delta-descriptors-change-based-place","slug":"delta-descriptors-change-based-place","title":"Delta Descriptors: Change-Based Place Representation for Robust Visual Localization","date":"2020-06-10","arxiv_id":"2006.05700","repositories_listed":1,"syntology":null},{"url":"/paper/screencast-tutorial-video-understanding","slug":"screencast-tutorial-video-understanding","title":"Screencast Tutorial Video Understanding","date":"2020-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/searching-for-actions-on-the-hyperbole","slug":"searching-for-actions-on-the-hyperbole","title":"Searching for Actions on the Hyperbole","date":"2020-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/video-playback-rate-perception-for-self","slug":"video-playback-rate-perception-for-self","title":"Video Playback Rate Perception for Self-Supervised Spatio-Temporal Representation Learning","date":"2020-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/condensed-movies-story-based-retrieval-with","slug":"condensed-movies-story-based-retrieval-with","title":"Condensed Movies: Story Based Retrieval with Contextual Embeddings","date":"2020-05-08","arxiv_id":"2005.04208","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/condensed-movies-story-based-retrieval-with#ran","syntology_url":"https://syntology.ai/paper/2005.04208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.04208"}},"official":null}},{"url":"/paper/speednet-learning-the-speediness-in-videos","slug":"speednet-learning-the-speediness-in-videos","title":"SpeedNet: Learning the Speediness in Videos","date":"2020-04-13","arxiv_id":"2004.06130","repositories_listed":1,"syntology":null},{"url":"/paper/amil-adversarial-multi-instance-learning-for","slug":"amil-adversarial-multi-instance-learning-for","title":"AMIL: Adversarial Multi Instance Learning for Human Pose Estimation","date":"2020-03-18","arxiv_id":"2003.08002","repositories_listed":1,"syntology":null},{"url":"/paper/noise-estimation-using-density-estimation-for","slug":"noise-estimation-using-density-estimation-for","title":"Noise Estimation Using Density Estimation for Self-Supervised Multimodal Learning","date":"2020-03-06","arxiv_id":"2003.03186","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-spatio-temporal-2","slug":"self-supervised-spatio-temporal-2","title":"Self-Supervised Visual Learning by Variable Playback Speeds Prediction of a Video","date":"2020-03-05","arxiv_id":"2003.02692","repositories_listed":1,"syntology":null},{"url":"/paper/video-cloze-procedure-for-self-supervised","slug":"video-cloze-procedure-for-self-supervised","title":"Video Cloze Procedure for Self-Supervised Spatio-Temporal Learning","date":"2020-01-02","arxiv_id":"2001.00294","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-cloze-procedure-for-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2001.00294","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.00294"}},"official":null}},{"url":"/paper/multimedia-retrieval-through-unsupervised","slug":"multimedia-retrieval-through-unsupervised","title":"Multimedia Retrieval Through Unsupervised Hypergraph-Based Manifold Ranking","date":"2019-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/visil-fine-grained-spatio-temporal-video","slug":"visil-fine-grained-spatio-temporal-video","title":"ViSiL: Fine-grained Spatio-Temporal Video Similarity Learning","date":"2019-08-20","arxiv_id":"1908.07410","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/visil-fine-grained-spatio-temporal-video#ran","syntology_url":"https://syntology.ai/paper/1908.07410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.07410"}},"official":{"repos":["MKLab-ITI/visil"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/central-similarity-hashing-via-hadamard","slug":"central-similarity-hashing-via-hadamard","title":"Central Similarity Quantization for Efficient Image and Video Retrieval","date":"2019-08-01","arxiv_id":"1908.00347","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/central-similarity-hashing-via-hadamard#ran","syntology_url":"https://syntology.ai/paper/1908.00347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.00347"}},"official":{"repos":["yuanli2333/Hadamard-Matrix-for-hashing"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-dense-encoding-for-zero-example-video","slug":"dual-dense-encoding-for-zero-example-video","title":"Dual Encoding for Zero-Example Video Retrieval","date":"2018-09-17","arxiv_id":"1809.06181","repositories_listed":1,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":2,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dual-dense-encoding-for-zero-example-video#ran","syntology_url":"https://syntology.ai/paper/1809.06181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.06181"}},"official":{"repos":["danieljf24/dual_encoding"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fivr-fine-grained-incident-video-retrieval","slug":"fivr-fine-grained-incident-video-retrieval","title":"FIVR: Fine-grained Incident Video Retrieval","date":"2018-09-11","arxiv_id":"1809.04094","repositories_listed":1,"syntology":null},{"url":"/paper/target-image-video-search-based-on-local","slug":"target-image-video-search-based-on-local","title":"Video Logo Retrieval based on local Features","date":"2018-08-11","arxiv_id":"1808.03735","repositories_listed":1,"syntology":null},{"url":"/paper/talking-face-generation-by-adversarially","slug":"talking-face-generation-by-adversarially","title":"Talking Face Generation by Adversarially Disentangled Audio-Visual Representation","date":"2018-07-20","arxiv_id":"1807.07860","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/talking-face-generation-by-adversarially#ran","syntology_url":"https://syntology.ai/paper/1807.07860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.07860"}},"official":null}},{"url":"/paper/learning-joint-embedding-with-multimodal-cues","slug":"learning-joint-embedding-with-multimodal-cues","title":"Learning Joint Embedding with Multimodal Cues for Cross-Modal Video-Text Retrieval","date":"2018-06-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/lamv-learning-to-align-and-match-videos-with","slug":"lamv-learning-to-align-and-match-videos-with","title":"LAMV: Learning to Align and Match Videos With Kernelized Temporal Layers","date":"2018-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-hashing-with-category-mask-for-fast","slug":"deep-hashing-with-category-mask-for-fast","title":"Deep Hashing with Category Mask for Fast Video Retrieval","date":"2017-12-22","arxiv_id":"1712.08315","repositories_listed":1,"syntology":null},{"url":"/paper/circulant-temporal-encoding-for-video","slug":"circulant-temporal-encoding-for-video","title":"Circulant temporal encoding for video retrieval and temporal alignment","date":"2015-06-08","arxiv_id":"1506.02588","repositories_listed":1,"syntology":null},{"url":null,"slug":"magmar-shared-task-system-description-video","title":"MAGMaR Shared Task System Description: Video Retrieval with OmniEmbed","date":"2025-06-11","arxiv_id":"2506.09409","repositories_listed":0,"syntology":null},{"url":null,"slug":"q2e-query-to-event-decomposition-for-zero","title":"Q2E: Query-to-Event Decomposition for Zero-Shot Multilingual Text-to-Video Retrieval","date":"2025-06-11","arxiv_id":"2506.10202","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-auxiliary-information-in-text-to","title":"Leveraging Auxiliary Information in Text-to-Video Retrieval: A Review","date":"2025-05-29","arxiv_id":"2505.23952","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-world-models-for-interactive-video","title":"Learning World Models for Interactive Video Generation","date":"2025-05-28","arxiv_id":"2505.21996","repositories_listed":0,"syntology":null},{"url":null,"slug":"cmawrnet-multiple-adverse-weather-removal-via","title":"CMAWRNet: Multiple Adverse Weather Removal via a Unified Quaternion Neural Architecture","date":"2025-05-03","arxiv_id":"2505.01882","repositories_listed":0,"syntology":null},{"url":null,"slug":"empowering-agentic-video-analytics-systems","title":"Empowering Agentic Video Analytics Systems with Video Language Models","date":"2025-05-01","arxiv_id":"2505.00254","repositories_listed":0,"syntology":null},{"url":null,"slug":"prototypes-are-balanced-units-for-efficient","title":"Prototypes are Balanced Units for Efficient and Effective Partially Relevant Video Retrieval","date":"2025-04-17","arxiv_id":"2504.13035","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-and-robust-moment-retrieval","title":"Towards Efficient and Robust Moment Retrieval System: A Unified Framework for Multi-Granularity Models and Temporal Reranking","date":"2025-04-11","arxiv_id":"2504.08384","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-modality-tags-for-enhanced-cross","title":"Leveraging Modality Tags for Enhanced Cross-Modal Video Retrieval","date":"2025-04-02","arxiv_id":"2504.01591","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-colbert-contextualized-late-interaction-1","title":"Video-ColBERT: Contextualized Late Interaction for Text-to-Video Retrieval","date":"2025-03-24","arxiv_id":"2503.19009","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-vmnet-accelerating-long-form-video","title":"Long-VMNet: Accelerating Long-Form Video Understanding via Fixed Memory","date":"2025-03-17","arxiv_id":"2503.13707","repositories_listed":0,"syntology":null},{"url":null,"slug":"quality-over-quantity-llm-based-curation-for","title":"Quality Over Quantity? LLM-Based Curation for a Data-Efficient Audio-Video Foundation Model","date":"2025-03-12","arxiv_id":"2503.09205","repositories_listed":0,"syntology":null},{"url":null,"slug":"narrating-the-video-boosting-text-video","title":"Narrating the Video: Boosting Text-Video Retrieval via Comprehensive Utilization of Frame-Level Captions","date":"2025-03-07","arxiv_id":"2503.05186","repositories_listed":0,"syntology":null},{"url":null,"slug":"llave-large-language-and-vision-embedding","title":"LLaVE: Large Language and Vision Embedding Models with Hardness-Weighted Contrastive Learning","date":"2025-03-04","arxiv_id":"2503.04812","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-generate-long-term-future","title":"Learning to Generate Long-term Future Narrations Describing Activities of Daily Living","date":"2025-03-03","arxiv_id":"2503.01416","repositories_listed":0,"syntology":null},{"url":null,"slug":"transmamba-fast-universal-architecture","title":"TransMamba: Fast Universal Architecture Adaption from Transformers to Mamba","date":"2025-02-21","arxiv_id":"2502.15130","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-ghost-investigating-ranking-bias","title":"Generative Ghost: Investigating Ranking Bias Hidden in AI-Generated Videos","date":"2025-02-11","arxiv_id":"2502.07327","repositories_listed":0,"syntology":null},{"url":null,"slug":"horus-multimodal-large-language-models","title":"HORUS: Multimodal Large Language Models Framework for Video Retrieval at VBS 2025","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-video-text-retrieval-a-new","title":"CaReBench: A Fine-Grained Benchmark for Video Captioning and Retrieval","date":"2024-12-31","arxiv_id":"2501.00513","repositories_listed":0,"syntology":null},{"url":null,"slug":"polysmart-trecvid-2024-medical-video-question","title":"PolySmart @ TRECVid 2024 Medical Video Question Answering","date":"2024-12-20","arxiv_id":"2412.15514","repositories_listed":0,"syntology":null},{"url":null,"slug":"query-centric-audio-visual-cognition-network","title":"Query-centric Audio-Visual Cognition Network for Moment Retrieval, Segmentation and Step-Captioning","date":"2024-12-18","arxiv_id":"2412.13543","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-semantic-communication","title":"Generative Semantic Communication: Architectures, Technologies, and Applications","date":"2024-12-11","arxiv_id":"2412.08642","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-contextualized-support-for","title":"Multimodal Contextualized Support for Enhancing Video Retrieval System","date":"2024-12-10","arxiv_id":"2412.07584","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-signed-language-instructions-in","title":"Generating Signed Language Instructions in Large-Scale Dialogue Systems","date":"2024-10-17","arxiv_id":"2410.14026","repositories_listed":0,"syntology":null},{"url":null,"slug":"multivent-2-0-a-massive-multilingual","title":"MultiVENT 2.0: A Massive Multilingual Benchmark for Event-Centric Video Retrieval","date":"2024-10-15","arxiv_id":"2410.11619","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoclip-xl-advancing-long-description","title":"VideoCLIP-XL: Advancing Long Description Understanding for Video CLIP Models","date":"2024-10-01","arxiv_id":"2410.00741","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-dataflywheel-resolving-the-impossible","title":"Video DataFlywheel: Resolving the Impossible Data Trinity in Video-Language Understanding","date":"2024-09-29","arxiv_id":"2409.19532","repositories_listed":0,"syntology":null},{"url":null,"slug":"unfolding-videos-dynamics-via-taylor","title":"Unfolding Videos Dynamics via Taylor Expansion","date":"2024-09-04","arxiv_id":"2409.02371","repositories_listed":0,"syntology":null},{"url":null,"slug":"sync-from-the-sea-retrieving-alignable-videos","title":"Sync from the Sea: Retrieving Alignable Videos from Large-Scale Datasets","date":"2024-09-02","arxiv_id":"2409.01445","repositories_listed":0,"syntology":null},{"url":null,"slug":"navero-unlocking-fine-grained-semantics-for","title":"NAVERO: Unlocking Fine-Grained Semantics for Video-Language Compositionality","date":"2024-08-18","arxiv_id":"2408.09511","repositories_listed":0,"syntology":null},{"url":null,"slug":"gqe-generalized-query-expansion-for-enhanced","title":"Bridging Information Asymmetry in Text-video Retrieval: A Data-centric Approach","date":"2024-08-14","arxiv_id":"2408.07249","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02672","title":"Latent-INR: A Flexible Framework for Implicit Representations of Videos with Discriminative Semantics","date":"2024-08-05","arxiv_id":"2408.02672","repositories_listed":0,"syntology":null},{"url":null,"slug":"expertaf-expert-actionable-feedback-from","title":"ExpertAF: Expert Actionable Feedback from Video","date":"2024-08-01","arxiv_id":"2408.00672","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-graph-matching-for-video-retrieval-in","title":"Neural Graph Matching for Video Retrieval in Large-Scale Video-driven E-commerce","date":"2024-08-01","arxiv_id":"2408.00346","repositories_listed":0,"syntology":null},{"url":null,"slug":"not-all-pairs-are-equal-hierarchical-learning","title":"Not All Pairs are Equal: Hierarchical Learning for Average-Precision-Oriented Video Retrieval","date":"2024-07-22","arxiv_id":"2407.15566","repositories_listed":0,"syntology":null},{"url":null,"slug":"merlin-multimodal-embedding-refinement-via","title":"MERLIN: Multimodal Embedding Refinement via LLM-based Iterative Navigation for Text-Video Retrieval-Rerank Pipeline","date":"2024-07-17","arxiv_id":"2407.12508","repositories_listed":0,"syntology":null},{"url":null,"slug":"ea-vtr-event-aware-video-text-retrieval","title":"EA-VTR: Event-Aware Video-Text Retrieval","date":"2024-07-10","arxiv_id":"2407.07478","repositories_listed":0,"syntology":null},{"url":null,"slug":"ace-a-generative-cross-modal-retrieval","title":"ACE: A Generative Cross-Modal Retrieval Framework with Coarse-To-Fine Semantic Modeling","date":"2024-06-25","arxiv_id":"2406.17507","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-granularity-and-multi-modal-feature","title":"Multi-Granularity and Multi-modal Feature Interaction Approach for Text Video Retrieval","date":"2024-06-21","arxiv_id":"2407.12798","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-holistic-language-video","title":"Towards Holistic Language-video Representation: the language model-enhanced MSR-Video to Text Dataset","date":"2024-06-19","arxiv_id":"2406.13809","repositories_listed":0,"syntology":null},{"url":null,"slug":"rnns-cnns-and-transformers-in-human-action","title":"RNNs, CNNs and Transformers in Human Action Recognition: A Survey and a Hybrid Model","date":"2024-06-02","arxiv_id":"2407.06162","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-sign-language-video","title":"Uncertainty-aware sign language video retrieval with probability distribution modeling","date":"2024-05-30","arxiv_id":"2405.19689","repositories_listed":0,"syntology":null},{"url":null,"slug":"rap-efficient-text-video-retrieval-with","title":"RAP: Efficient Text-Video Retrieval with Sparse-and-Correlated Adapter","date":"2024-05-29","arxiv_id":"2405.19465","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-interactive-image-retrieval-with","title":"Enhancing Interactive Image Retrieval With Query Rewriting Using Large Language Models and Vision Language Models","date":"2024-04-29","arxiv_id":"2404.18746","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-text-to-video-retrieval-from-image","title":"Learning text-to-video retrieval from image captioning","date":"2024-04-26","arxiv_id":"2404.17498","repositories_listed":0,"syntology":null}],"record_sha256":"04c4106efe313c3b477a595af082fc5283a02faafd45d9650518aa4a6e285516","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}