{"url":"/dataset/qvhighlights","name":"QVHighlights","full_name":"Query-based Video Highlights","description_markdown":"The Query-based Video Highlights (**QVHighlights**) dataset is a dataset for detecting customized moments and highlights from videos given natural language (NL). It consists of over 10,000 YouTube videos, covering a wide range of topics, from everyday activities and travel in lifestyle vlog videos to social and political activities in news videos. Each video in the dataset is annotated with: (1) a human-written free-form NL query, (2) relevant moments in the video w.r.t. the query, and (3) five-point scale saliency scores for all query-relevant clips.","description_withheld":null,"homepage":"https://github.com/jayleicn/moment_detr/tree/main/data","introduced_date":"2021-07-20","introduced_date_note":null,"introduced_by":{"paper":"/paper/qvhighlights-detecting-moments-and-highlights","title":"QVHighlights: Detecting Moments and Highlights in Videos via Natural Language Queries","first_author":"Jie Lei","url":null},"license":{"name":"Attribution-NonCommercial-ShareAlike 4.0 International","url":"https://github.com/jayleicn/moment_detr/blob/main/data/LICENSE"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Highlight Detection","url":"/task/highlight-detection","datasets_with_task":"/datasets/task/highlight-detection"},{"name":"Moment Retrieval","url":"/task/moment-retrieval","datasets_with_task":"/datasets/task/moment-retrieval"},{"name":"Video Grounding","url":"/task/video-grounding","datasets_with_task":"/datasets/task/video-grounding"},{"name":"Zero-shot Moment Retrieval","url":"/task/zero-shot-moment-retrieval","datasets_with_task":"/datasets/task/zero-shot-moment-retrieval"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["QVHighlights"],"data_loaders":[{"repo":"https://github.com/jayleicn/moment_detr","url":"https://github.com/jayleicn/moment_detr","frameworks":["pytorch"]}],"num_papers_in_archive":41,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/moment-retrieval-on-qvhighlights","task":"Moment Retrieval","dataset_variant":"QVHighlights","rows":32,"metrics":["mAP","R@1 IoU=0.5","R@1 IoU=0.7","mAP@0.5","mAP@0.75"],"first_row_in_archive_order":{"model":"SG-DETR (w/ PT)","paper":"/paper/saliency-guided-detr-for-moment-retrieval-and","metrics":{"R@1 IoU=0.5":"74.20","R@1 IoU=0.7":"60.40","mAP":"58.80","mAP@0.5":"76.20","mAP@0.75":"60.80"},"code_links":[{"title":"ai-forever/sg-detr","url":"https://github.com/ai-forever/sg-detr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/highlight-detection-on-qvhighlights","task":"Highlight Detection","dataset_variant":"QVHighlights","rows":21,"metrics":["mAP","Hit@1"],"first_row_in_archive_order":{"model":"SG-DETR (w/ PT)","paper":"/paper/saliency-guided-detr-for-moment-retrieval-and","metrics":{"Hit@1":"71.00","mAP":"44.70"},"code_links":[{"title":"ai-forever/sg-detr","url":"https://github.com/ai-forever/sg-detr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-grounding-on-qvhighlights","task":"Video Grounding","dataset_variant":"QVHighlights","rows":7,"metrics":["R@1,IoU=0.7","R@1,IoU=0.5"],"first_row_in_archive_order":{"model":"InternVideo2-6B","paper":"/paper/internvideo2-scaling-video-foundation-models","metrics":{"R@1,IoU=0.5":"71.42","R@1,IoU=0.7":"56.45"},"code_links":[{"title":"opengvlab/internvideo","url":"https://github.com/opengvlab/internvideo"},{"title":"opengvlab/internvideo2","url":"https://github.com/opengvlab/internvideo2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-moment-retrieval-on-qvhighlights","task":"Zero-shot Moment Retrieval","dataset_variant":"QVHighlights","rows":2,"metrics":["R1@0.5","R1@0.7","mAP","mAP@0.5","mAP@0.75"],"first_row_in_archive_order":{"model":"SG-DETR (ZS)","paper":"/paper/saliency-guided-detr-for-moment-retrieval-and","metrics":{"R1@0.5":"63.90","R1@0.7":"49.60","mAP":"48.30","mAP@0.5":"67.50","mAP@0.75":"49.00"},"code_links":[{"title":"ai-forever/sg-detr","url":"https://github.com/ai-forever/sg-detr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/ld-detr-loop-decoder-detection-transformer","title":"LD-DETR: Loop Decoder DEtection TRansformer for Video Moment Retrieval and Highlight Detection","date":"2025-01-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/length-aware-detr-for-robust-moment-retrieval","title":"Length-Aware DETR for Robust Moment Retrieval","date":"2024-12-30","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/flashvtg-feature-layering-and-adaptive-score","title":"FlashVTG: Feature Layering and Adaptive Score Handling Network for Video Temporal Grounding","date":"2024-12-18","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":2,"samples_unverified":11,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videolights-feature-refinement-and-cross-task","title":"VideoLights: Feature Refinement and Cross-Task Alignment Transformer for Joint Video Highlight Detection and Moment Retrieval","date":"2024-12-02","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/llava-mr-large-language-and-vision-assistant","title":"LLaVA-MR: Large Language-and-Vision Assistant for Video Moment Retrieval","date":"2024-11-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/number-it-temporal-grounding-videos-like","title":"Number it: Temporal Grounding Videos like Flipping Manga","date":"2024-11-15","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":1,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/timesuite-improving-mllms-for-long-video","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","date":"2024-10-25","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/saliency-guided-detr-for-moment-retrieval-and","title":"Saliency-Guided DETR for Moment Retrieval and Highlight Detection","date":"2024-10-02","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/prior-knowledge-integration-via-llm-encoding","title":"Prior Knowledge Integration via LLM Encoding and Pseudo Event Regulation for Video Moment Retrieval","date":"2024-07-21","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unleash-the-potential-of-clip-for-video","title":"Unleash the Potential of CLIP for Video Highlight Detection","date":"2024-04-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/r-2-tuning-efficient-image-to-video-transfer-1","title":"$R^2$-Tuning: Efficient Image-to-Video Transfer Learning for Video Temporal Grounding","date":"2024-03-31","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/video-mamba-suite-state-space-model-as-a","title":"Video Mamba Suite: State Space Model as a Versatile Alternative for Video Understanding","date":"2024-03-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":8,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vtg-gpt-tuning-free-zero-shot-video-temporal-1","title":"VTG-GPT: Tuning-Free Zero-Shot Video Temporal Grounding with GPT","date":"2024-03-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":9,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bam-detr-boundary-aligned-moment-detection","title":"BAM-DETR: Boundary-Aligned Moment Detection Transformer for Temporal Sentence Grounding in Videos","date":"2023-11-30","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":8,"samples_unverified":3,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bridging-the-gap-a-unified-video","title":"Bridging the Gap: A Unified Video Comprehension Framework for Moment Retrieval and Highlight Detection","date":"2023-11-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":9,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/correlation-guided-query-dependency","title":"Correlation-Guided Query-Dependency Calibration for Video Temporal Grounding","date":"2023-11-15","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":2,"samples_unverified":9,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/diffusionvmr-diffusion-model-for-video-moment","title":"DiffusionVMR: Diffusion Model for Joint Video Moment Retrieval and Highlight Detection","date":"2023-08-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/unloc-a-unified-framework-for-video","title":"UnLoc: A Unified Framework for Video Localization Tasks","date":"2023-08-21","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/univtg-towards-unified-video-language","title":"UniVTG: Towards Unified Video-Language Temporal Grounding","date":"2023-07-31","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":10,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/overcoming-weak-visual-textual-alignment-for","title":"Background-aware Moment Detection for Video Moment Retrieval","date":"2023-06-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/boundary-denoising-for-video-activity","title":"Boundary-Denoising for Video Activity Localization","date":"2023-04-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/query-dependent-video-representation-for","title":"Query-Dependent Video Representation for Moment Retrieval and Highlight Detection","date":"2023-03-24","rows_on_this_dataset":9,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/umt-unified-multi-modal-transformers-for","title":"UMT: Unified Multi-modal Transformers for Joint Video Moment Retrieval and Highlight Detection","date":"2022-03-23","rows_on_this_dataset":5,"code_links":3,"syntology":null},{"paper":"/paper/detecting-moments-and-highlights-in-videos","title":"Detecting Moments and Highlights in Videos via Natural Language Queries","date":"2021-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/qvhighlights-detecting-moments-and-highlights","title":"QVHighlights: Detecting Moments and Highlights in Videos via Natural Language Queries","date":"2021-07-20","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":13,"samples_harvested":124,"samples_ran":63,"samples_unverified":61,"pointer_only_for_licence":37,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}