{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-retrieval/papers/4","list_of":"/task/video-retrieval","task":"Video Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":5,"rows_per_page":100,"rows":[301,400],"of":486,"counts":{"archive_papers_tagged":486,"with_a_code_link":255,"where_syntology_ran_a_sample":95,"not_listed_spam_title":0,"listed":486,"listed_where_code_ran":95,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":76,"every_run_a_failure_of_syntologys_instrument":19,"listed_with_a_run_with_no_instrument_failure":76,"listed_every_run_a_failure_of_syntologys_instrument":19,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-retrieval","prev":"/task/video-retrieval/papers/3","next":"/task/video-retrieval/papers/5","papers":[{"url":null,"slug":"she-net-syntax-hierarchy-enhanced-text-video","title":"SHE-Net: Syntax-Hierarchy-Enhanced Text-Video Retrieval","date":"2024-04-22","arxiv_id":"2404.14066","repositories_listed":0,"syntology":null},{"url":null,"slug":"prota-probabilistic-token-aggregation-for","title":"ProTA: Probabilistic Token Aggregation for Text-Video Retrieval","date":"2024-04-18","arxiv_id":"2404.12216","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-is-mass-modeling-as-stochastic-embedding","title":"Text Is MASS: Modeling as Stochastic Embedding for Text-Video Retrieval","date":"2024-03-26","arxiv_id":"2403.17998","repositories_listed":0,"syntology":null},{"url":null,"slug":"event-aware-video-corpus-moment-retrieval","title":"Event-aware Video Corpus Moment Retrieval","date":"2024-02-21","arxiv_id":"2402.13566","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-editing-for-video-retrieval","title":"Video Editing for Video Retrieval","date":"2024-02-04","arxiv_id":"2402.02335","repositories_listed":0,"syntology":null},{"url":null,"slug":"coavt-a-cognition-inspired-unified-audio","title":"CoAVT: A Cognition-Inspired Unified Audio-Visual-Text Pre-Training Model for Multimodal Processing","date":"2024-01-22","arxiv_id":"2401.12264","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-vision-language-models-on-millions","title":"Distilling Vision-Language Models on Millions of Videos","date":"2024-01-11","arxiv_id":"2401.06129","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-video-retrieval-via-variational-multi","title":"Text-Video Retrieval via Variational Multi-Modal Hypergraph Networks","date":"2024-01-06","arxiv_id":"2401.03177","repositories_listed":0,"syntology":null},{"url":null,"slug":"detours-for-navigating-instructional-videos","title":"Detours for Navigating Instructional Videos","date":"2024-01-03","arxiv_id":"2401.01823","repositories_listed":0,"syntology":null},{"url":null,"slug":"no-more-shortcuts-realizing-the-potential-of","title":"No More Shortcuts: Realizing the Potential of Temporal Self-Supervision","date":"2023-12-20","arxiv_id":"2312.13008","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-generative-language-models-for","title":"Leveraging Generative Language Models for Weakly Supervised Sentence Component Analysis in Video-Language Joint Learning","date":"2023-12-10","arxiv_id":"2312.06699","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-vision-language-models-learn","title":"Vision-Language Models Learn Super Images for Efficient Partially Relevant Video Retrieval","date":"2023-12-01","arxiv_id":"2312.00414","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-video-is-worth-10000-words-training-and","title":"A Video is Worth 10,000 Words: Training and Benchmarking with Diverse Captions for Better Long Video Retrieval","date":"2023-11-30","arxiv_id":"2312.00115","repositories_listed":0,"syntology":null},{"url":null,"slug":"spacewalk-18-a-benchmark-for-multimodal-and","title":"Spacewalk-18: A Benchmark for Multimodal and Long-form Procedural Video Understanding","date":"2023-11-30","arxiv_id":"2311.18773","repositories_listed":0,"syntology":null},{"url":null,"slug":"e-vilm-efficient-video-language-model-via","title":"E-ViLM: Efficient Video-Language Model via Masked Video Modeling with Semantic Vector-Quantized Tokenizer","date":"2023-11-28","arxiv_id":"2311.17267","repositories_listed":0,"syntology":null},{"url":null,"slug":"sinkhorn-transformations-for-single-query","title":"Sinkhorn Transformations for Single-Query Postprocessing in Text-Video Retrieval","date":"2023-11-14","arxiv_id":"2311.08143","repositories_listed":0,"syntology":null},{"url":"/paper/lost-your-style-navigating-with-semantic","slug":"lost-your-style-navigating-with-semantic","title":"Lost Your Style? Navigating with Semantic-Level Approach for Text-to-Outfit Retrieval","date":"2023-11-03","arxiv_id":"2311.02122","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-study-of-frame-selection-for","title":"An Empirical Study of Frame Selection for Text-to-Video Retrieval","date":"2023-11-01","arxiv_id":"2311.00298","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-exploring-global-local-spatio-temporal","title":"CHAIN: Exploring Global-Local Spatio-Temporal Information for Improved Self-Supervised Video Hashing","date":"2023-10-29","arxiv_id":"2310.18926","repositories_listed":0,"syntology":null},{"url":"/paper/videoprompter-an-ensemble-of-foundational","slug":"videoprompter-an-ensemble-of-foundational","title":"Videoprompter: an ensemble of foundational models for zero-shot video understanding","date":"2023-10-23","arxiv_id":"2310.15324","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-stream-knowledge-preserving-hashing-for","title":"Dual-Stream Knowledge-Preserving Hashing for Unsupervised Video Retrieval","date":"2023-10-12","arxiv_id":"2310.08009","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-segment-similarity-and-alignment-in","title":"Learning Segment Similarity and Alignment in Large-Scale Content Based Video Retrieval","date":"2023-09-20","arxiv_id":"2309.11091","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-debiasing-frame-length-bias-in-text","title":"Towards Debiasing Frame Length Bias in Text-Video Retrieval via Causal Intervention","date":"2023-09-17","arxiv_id":"2309.09311","repositories_listed":0,"syntology":null},{"url":null,"slug":"teachclip-multi-grained-teaching-for","title":"TeachCLIP: Multi-Grained Teaching for Efficient Text-to-Video Retrieval","date":"2023-08-02","arxiv_id":"2308.01217","repositories_listed":0,"syntology":null},{"url":"/paper/audio-enhanced-text-to-video-retrieval-using","slug":"audio-enhanced-text-to-video-retrieval-using","title":"Audio-Enhanced Text-to-Video Retrieval using Text-Conditioned Feature Alignment","date":"2023-07-24","arxiv_id":"2307.12964","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-text-video-retrieval-with-frozen","title":"Fine-grained Text-Video Retrieval with Frozen Image Encoders","date":"2023-07-14","arxiv_id":"2307.09972","repositories_listed":0,"syntology":null},{"url":null,"slug":"multivent-multilingual-videos-of-events-with","title":"MultiVENT: Multilingual Videos of Events with Aligned Natural Text","date":"2023-07-06","arxiv_id":"2307.03153","repositories_listed":0,"syntology":null},{"url":null,"slug":"key-frame-extraction-with-attention-based","title":"Key Frame Extraction with Attention Based Deep Neural Networks","date":"2023-06-21","arxiv_id":"2306.13176","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-multimodal-representation-learning-1","title":"Enhanced Multimodal Representation Learning with Cross-modal KD","date":"2023-06-13","arxiv_id":"2306.07646","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-of-challenges-in-egocentric-text","title":"An Overview of Challenges in Egocentric Text-Video Retrieval","date":"2023-06-07","arxiv_id":"2306.04345","repositories_listed":0,"syntology":null},{"url":null,"slug":"marinevrs-marine-video-retrieval-system-with","title":"MarineVRS: Marine Video Retrieval System with Explainability via Semantic Understanding","date":"2023-06-07","arxiv_id":"2306.04593","repositories_listed":0,"syntology":null},{"url":null,"slug":"fpgahart-a-toolflow-for-throughput-oriented","title":"fpgaHART: A toolflow for throughput-oriented acceleration of 3D CNNs for HAR onto FPGAs","date":"2023-05-31","arxiv_id":"2305.19896","repositories_listed":0,"syntology":null},{"url":null,"slug":"fmm-x3d-fpga-based-modeling-and-mapping-of","title":"FMM-X3D: FPGA-based modeling and mapping of X3D for Human Action Recognition","date":"2023-05-29","arxiv_id":"2305.18479","repositories_listed":0,"syntology":null},{"url":"/paper/vlab-enhancing-video-language-pre-training-by","slug":"vlab-enhancing-video-language-pre-training-by","title":"VLAB: Enhancing Video Language Pre-training by Feature Adapting and Blending","date":"2023-05-22","arxiv_id":"2305.13167","repositories_listed":0,"syntology":null},{"url":null,"slug":"mask-to-reconstruct-cooperative-semantics","title":"Mask to reconstruct: Cooperative Semantics Completion for Video-text Retrieval","date":"2023-05-13","arxiv_id":"2305.07910","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-deep-learning-for-video","title":"A Review of Deep Learning for Video Captioning","date":"2023-04-22","arxiv_id":"2304.11431","repositories_listed":0,"syntology":null},{"url":null,"slug":"laser-neuro-symbolic-learning-of-semantic","title":"LASER: A Neuro-Symbolic Framework for Learning Spatial-Temporal Scene Graphs with Weak Supervision","date":"2023-04-15","arxiv_id":"2304.07647","repositories_listed":0,"syntology":null},{"url":null,"slug":"free-form-multi-modal-multimedia-retrieval","title":"Free-Form Multi-Modal Multimedia Retrieval (4MR)","date":"2023-03-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"perfect-match-in-video-retrieval","title":"Perfect Match in Video Retrieval","date":"2023-03-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"colo-scrl-self-supervised-contrastive","title":"Colo-SCRL: Self-Supervised Contrastive Representation Learning for Colonoscopic Video Retrieval","date":"2023-03-28","arxiv_id":"2303.15671","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatiotemporally-discriminative-video","title":"Structured Video-Language Modeling with Temporal Grouping and Spatial Grounding","date":"2023-03-28","arxiv_id":"2303.16341","repositories_listed":0,"syntology":null},{"url":"/paper/multi-efficient-video-and-language","slug":"multi-efficient-video-and-language","title":"MuLTI: Efficient Video-and-Language Understanding with Text-Guided MultiWay-Sampler and Multiple Choice Modeling","date":"2023-03-10","arxiv_id":"2303.05707","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-video-retrieval-by-adaptive-margin","title":"Improving Video Retrieval by Adaptive Margin","date":"2023-03-09","arxiv_id":"2303.05093","repositories_listed":0,"syntology":null},{"url":null,"slug":"stoa-vlp-spatial-temporal-modeling-of-object","title":"STOA-VLP: Spatial-Temporal Modeling of Object and Action for Video-Language Pre-training","date":"2023-02-20","arxiv_id":"2302.09736","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-perceiving-video-language-pre","title":"Temporal Perceiving Video-Language Pre-training","date":"2023-01-18","arxiv_id":"2301.07463","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-trajectory-word-alignments-for-video","title":"Learning Trajectory-Word Alignments for Video-Language Tasks","date":"2023-01-05","arxiv_id":"2301.01953","repositories_listed":0,"syntology":null},{"url":null,"slug":"hivlp-hierarchical-interactive-video-language","title":"HiVLP: Hierarchical Interactive Video-Language Pre-Training","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/pidro-parallel-isomeric-attention-with","slug":"pidro-parallel-isomeric-attention-with","title":"PIDRo: Parallel Isomeric Attention with Dynamic Routing for Text-Video Retrieval","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/hitea-hierarchical-temporal-aware-video","slug":"hitea-hierarchical-temporal-aware-video","title":"HiTeA: Hierarchical Temporal-Aware Video-Language Pre-training","date":"2022-12-30","arxiv_id":"2212.14546","repositories_listed":0,"syntology":null},{"url":"/paper/video-text-modeling-with-zero-shot-transfer","slug":"video-text-modeling-with-zero-shot-transfer","title":"VideoCoCa: Video-Text Modeling with Zero-Shot Transfer from Contrastive Captioners","date":"2022-12-09","arxiv_id":"2212.04979","repositories_listed":0,"syntology":null},{"url":"/paper/masked-contrastive-pre-training-for-efficient","slug":"masked-contrastive-pre-training-for-efficient","title":"Masked Contrastive Pre-Training for Efficient Video-Text Retrieval","date":"2022-12-02","arxiv_id":"2212.00986","repositories_listed":0,"syntology":null},{"url":null,"slug":"renmin-university-of-china-at-trecvid-2022","title":"Renmin University of China at TRECVID 2022: Improving Video Search by Feature Fusion and Negation Understanding","date":"2022-11-28","arxiv_id":"2211.15039","repositories_listed":0,"syntology":null},{"url":null,"slug":"smaug-sparse-masked-autoencoder-for-efficient","title":"SMAUG: Sparse Masked Autoencoder for Efficient Video-Language Pre-training","date":"2022-11-21","arxiv_id":"2211.11446","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-model-for-video-understanding-and","title":"A Unified Model for Video Understanding and Knowledge Embedding with Heterogeneous Knowledge Graph Dataset","date":"2022-11-19","arxiv_id":"2211.10624","repositories_listed":0,"syntology":null},{"url":null,"slug":"clop-video-and-language-pre-training-with","title":"CLOP: Video-and-Language Pre-Training with Knowledge Regularizations","date":"2022-11-07","arxiv_id":"2211.03314","repositories_listed":0,"syntology":null},{"url":null,"slug":"litevl-efficient-video-language-learning-with","title":"LiteVL: Efficient Video-Language Learning with Enhanced Spatial-Temporal Modeling","date":"2022-10-21","arxiv_id":"2210.11929","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-video-moments-retrieval-at-scale-a","title":"Semantic Video Moments Retrieval at Scale: A New Task and a Baseline","date":"2022-10-15","arxiv_id":"2210.08389","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-video-language-learning-with-fine","title":"Contrastive Video-Language Learning with Fine-grained Frame Sampling","date":"2022-10-10","arxiv_id":"2210.05039","repositories_listed":0,"syntology":null},{"url":null,"slug":"fighting-fire-with-fire-assessing-the","title":"Fighting FIRe with FIRE: Assessing the Validity of Text-to-Video Retrieval Benchmarks","date":"2022-10-10","arxiv_id":"2210.05038","repositories_listed":0,"syntology":null},{"url":null,"slug":"event-extraction-in-video-transcripts","title":"Event Extraction in Video Transcripts","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"text-adaptive-multiple-visual-prototype","title":"Text-Adaptive Multiple Visual Prototype Matching for Video-Text Retrieval","date":"2022-09-27","arxiv_id":"2209.13307","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-granularity-graph-pooling-for-video","title":"Multi-Granularity Graph Pooling for Video-based Person Re-Identification","date":"2022-09-23","arxiv_id":"2209.11584","repositories_listed":0,"syntology":null},{"url":null,"slug":"pose-aided-video-based-person-re","title":"Pose-Aided Video-based Person Re-Identification via Recurrent Graph Convolutional Network","date":"2022-09-23","arxiv_id":"2209.11582","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-automatic-data-annotation-system-for","title":"Semi-automatic Data Annotation System for Multi-Target Multi-Camera Vehicle Tracking","date":"2022-09-20","arxiv_id":"2209.09606","repositories_listed":0,"syntology":null},{"url":null,"slug":"tree-based-text-vision-bert-for-video-search","title":"Tree-based Text-Vision BERT for Video Search in Baidu Video Advertising","date":"2022-09-19","arxiv_id":"2209.08759","repositories_listed":0,"syntology":null},{"url":"/paper/omnivl-one-foundation-model-for-image","slug":"omnivl-one-foundation-model-for-image","title":"OmniVL:One Foundation Model for Image-Language and Video-Language Tasks","date":"2022-09-15","arxiv_id":"2209.07526","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-contrastive-learning-with-curriculum","title":"Temporal Contrastive Learning with Curriculum","date":"2022-09-02","arxiv_id":"2209.00760","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-video-retrieval-using-multilingual","title":"MuMUR : Multilingual Multimodal Universal Retrieval","date":"2022-08-24","arxiv_id":"2208.11553","repositories_listed":0,"syntology":null},{"url":null,"slug":"star-gnn-spatial-temporal-video","title":"STAR-GNN: Spatial-Temporal Video Representation for Content-based Retrieval","date":"2022-08-15","arxiv_id":"2208.06966","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-sensitive-contrastive-learning-for","title":"Motion Sensitive Contrastive Learning for Self-supervised Video Representation","date":"2022-08-12","arxiv_id":"2208.06105","repositories_listed":0,"syntology":null},{"url":null,"slug":"qsam-net-rain-streak-removal-by-quaternion","title":"QSAM-Net: Rain streak removal by quaternion neural network with self-attention module","date":"2022-08-08","arxiv_id":"2208.04346","repositories_listed":0,"syntology":null},{"url":"/paper/lat-latent-translation-with-cycle-consistency","slug":"lat-latent-translation-with-cycle-consistency","title":"LaT: Latent Translation with Cycle-Consistency for Video-Text Retrieval","date":"2022-07-11","arxiv_id":"2207.04858","repositories_listed":0,"syntology":null},{"url":"/paper/vrag-region-attention-graphs-for-content","slug":"vrag-region-attention-graphs-for-content","title":"VRAG: Region Attention Graphs for Content-Based Video Retrieval","date":"2022-05-18","arxiv_id":"2205.09068","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-video-based-action-quality","title":"A Survey of Video-based Action Quality Assessment","date":"2022-04-20","arxiv_id":"2204.09271","repositories_listed":0,"syntology":null},{"url":null,"slug":"modality-balanced-embedding-for-video","title":"Modality-Balanced Embedding for Video Retrieval","date":"2022-04-18","arxiv_id":"2204.08182","repositories_listed":0,"syntology":null},{"url":"/paper/cots-collaborative-two-stream-vision-language","slug":"cots-collaborative-two-stream-vision-language","title":"COTS: Collaborative Two-Stream Vision-Language Pre-Training Model for Cross-Modal Retrieval","date":"2022-04-15","arxiv_id":"2204.07441","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-representations-for-video","title":"Probabilistic Representations for Video Contrastive Learning","date":"2022-04-08","arxiv_id":"2204.03946","repositories_listed":0,"syntology":null},{"url":"/paper/hunyuan-tvr-for-text-video-retrivial","slug":"hunyuan-tvr-for-text-video-retrivial","title":"Tencent Text-Video Retrieval: Hierarchical Cross-Modal Interactions with Multi-Level Representations","date":"2022-04-07","arxiv_id":"2204.03382","repositories_listed":0,"syntology":null},{"url":"/paper/learning-audio-video-modalities-from-image","slug":"learning-audio-video-modalities-from-image","title":"Learning Audio-Video Modalities from Image Captions","date":"2022-04-01","arxiv_id":"2204.00679","repositories_listed":0,"syntology":null},{"url":null,"slug":"create-a-benchmark-for-chinese-short-video-1","title":"CREATE: A Benchmark for Chinese Short Video Retrieval and Title Generation","date":"2022-03-31","arxiv_id":"2203.16763","repositories_listed":0,"syntology":null},{"url":null,"slug":"controllable-augmentations-for-video","title":"Controllable Augmentations for Video Representation Learning","date":"2022-03-30","arxiv_id":"2203.16632","repositories_listed":0,"syntology":null},{"url":"/paper/mdmmt-2-multidomain-multimodal-transformer","slug":"mdmmt-2-multidomain-multimodal-transformer","title":"MDMMT-2: Multidomain Multimodal Transformer for Video Retrieval, One More Step Towards Generalization","date":"2022-03-14","arxiv_id":"2203.07086","repositories_listed":0,"syntology":null},{"url":null,"slug":"live-laparoscopic-video-retrieval-with","title":"Live Laparoscopic Video Retrieval with Compressed Uncertainty","date":"2022-03-08","arxiv_id":"2203.04301","repositories_listed":0,"syntology":null},{"url":null,"slug":"vscript-controllable-script-generation-with","title":"VScript: Controllable Script Generation with Visual Presentation","date":"2022-03-01","arxiv_id":"2203.00314","repositories_listed":0,"syntology":null},{"url":"/paper/newskvqa-knowledge-aware-news-video-question","slug":"newskvqa-knowledge-aware-news-video-question","title":"NEWSKVQA: Knowledge-Aware News Video Question Answering","date":"2022-02-08","arxiv_id":"2202.04015","repositories_listed":0,"syntology":null},{"url":"/paper/end-to-end-generative-pretraining-for","slug":"end-to-end-generative-pretraining-for","title":"End-to-end Generative Pretraining for Multimodal Video Captioning","date":"2022-01-20","arxiv_id":"2201.08264","repositories_listed":0,"syntology":null},{"url":null,"slug":"watch-less-and-uncover-more-could-navigation","title":"Watch Less and Uncover More: Could Navigation Tools Help Users Search and Explore Videos?","date":"2022-01-10","arxiv_id":"2201.03408","repositories_listed":0,"syntology":null},{"url":null,"slug":"sign-language-video-retrieval-with-free-form","title":"Sign Language Video Retrieval with Free-Form Textual Queries","date":"2022-01-07","arxiv_id":"2201.02495","repositories_listed":0,"syntology":null},{"url":null,"slug":"sound-and-visual-representation-learning-with","title":"Sound and Visual Representation Learning with Multiple Pretraining Tasks","date":"2022-01-04","arxiv_id":"2201.01046","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-transformer-based-video-hashing","title":"Vision Transformer Based Video Hashing Retrieval for Tracing the Source of Fake Videos","date":"2021-12-15","arxiv_id":"2112.08117","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-spatiotemporal-representation","title":"Self-supervised Spatiotemporal Representation Learning by Exploiting Video Continuity","date":"2021-12-11","arxiv_id":"2112.05883","repositories_listed":0,"syntology":null},{"url":null,"slug":"stc-mix-space-time-channel-mixing-for-self","title":"Cross-modal Manifold Cutmix for Self-supervised Video Representation Learning","date":"2021-12-07","arxiv_id":"2112.03906","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-equivariant-contrastive-video-1","title":"Time-Equivariant Contrastive Video Representation Learning","date":"2021-12-07","arxiv_id":"2112.03624","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizable-multi-linear-attention-network","title":"Generalizable Multi-linear Attention Network","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"induce-edit-retrieve-language-grounded","title":"Induce, Edit, Retrieve:Language Grounded Multimodal Schema for Instructional Video Retrieval","date":"2021-11-17","arxiv_id":"2111.09276","repositories_listed":0,"syntology":null},{"url":null,"slug":"create-a-benchmark-for-chinese-short-video","title":"CREATE: A Benchmark for Chinese Short Video Retrieval and Title Generation","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/clip2tv-an-empirical-study-on-transformer","slug":"clip2tv-an-empirical-study-on-transformer","title":"CLIP2TV: Align, Match and Distill for Video-Text Retrieval","date":"2021-11-10","arxiv_id":"2111.05610","repositories_listed":0,"syntology":null},{"url":null,"slug":"swamp-swapped-assignment-of-multi-modal-pairs","title":"SwAMP: Swapped Assignment of Multi-Modal Pairs for Cross-Modal Retrieval","date":"2021-11-10","arxiv_id":"2111.05814","repositories_listed":0,"syntology":null},{"url":null,"slug":"masking-modalities-for-cross-modal-video","title":"Masking Modalities for Cross-modal Video Retrieval","date":"2021-11-01","arxiv_id":"2111.01300","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptation-in-multi-view-embedding-for","title":"Domain Adaptation in Multi-View Embedding for Cross-Modal Video Retrieval","date":"2021-10-25","arxiv_id":"2110.12812","repositories_listed":0,"syntology":null}],"record_sha256":"a02440430ea9889aadd0246d089dafb8eb0d5f195c1d20a162a4983750e695a0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}