{"url":"/sota/highlight-detection-on-qvhighlights","task":{"name":"Highlight Detection","url":"/task/highlight-detection","note":null},"dataset":{"name":"QVHighlights","url":"/dataset/qvhighlights"},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"https://youtu.be/pJ0auP7dbcY?si=vSiZevfJ57YUKC2q","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["mAP","Hit@1"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"mAP":"higher","Hit@1":null}},"counts":{"rows":21,"rows_with_code":20,"rows_with_paper_page":20,"rows_dated":20,"rows_using_additional_data":4},"rows":[{"rank_in_archive_order":1,"model":"SG-DETR (w/ PT)","metrics":{"Hit@1":"71.00","mAP":"44.70"},"uses_additional_data":true,"paper_date":"2024-10-02","paper":"/paper/saliency-guided-detr-for-moment-retrieval-and","paper_url":"https://arxiv.org/abs/2410.01615v1","paper_title":"Saliency-Guided DETR for Moment Retrieval and Highlight Detection","code":"https://github.com/ai-forever/sg-detr","n_code_links":1,"syntology":null},{"rank_in_archive_order":2,"model":"FlashVTG","metrics":{"Hit@1":"71.01","mAP":"44.09"},"uses_additional_data":false,"paper_date":"2024-12-18","paper":"/paper/flashvtg-feature-layering-and-adaptive-score","paper_url":"https://arxiv.org/abs/2412.13441v1","paper_title":"FlashVTG: Feature Layering and Adaptive Score Handling Network for Video Temporal Grounding","code":"https://github.com/zhuo-cao/flashvtg","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":11,"n_samples":13,"n_pointer_only_licence":13}},{"rank_in_archive_order":3,"model":"SG-DETR","metrics":{"Hit@1":"69.13","mAP":"43.76"},"uses_additional_data":false,"paper_date":"2024-10-02","paper":"/paper/saliency-guided-detr-for-moment-retrieval-and","paper_url":"https://arxiv.org/abs/2410.01615v1","paper_title":"Saliency-Guided DETR for Moment Retrieval and Highlight Detection","code":"https://github.com/ai-forever/sg-detr","n_code_links":1,"syntology":null},{"rank_in_archive_order":4,"model":"VideoLights-B-pt","metrics":{"Hit@1":"70.56","mAP":"42.84"},"uses_additional_data":true,"paper_date":"2024-12-02","paper":"/paper/videolights-feature-refinement-and-cross-task","paper_url":"https://arxiv.org/abs/2412.01558v1","paper_title":"VideoLights: Feature Refinement and Cross-Task Alignment Transformer for Joint Video Highlight Detection and Moment Retrieval","code":"https://github.com/dpaul06/VideoLights","n_code_links":1,"syntology":null},{"rank_in_archive_order":5,"model":"HL-CLIP","metrics":{"Hit@1":"70.60","mAP":"41.94"},"uses_additional_data":false,"paper_date":"2024-04-02","paper":"/paper/unleash-the-potential-of-clip-for-video","paper_url":"https://arxiv.org/abs/2404.01745v1","paper_title":"Unleash the Potential of CLIP for Video Highlight Detection","code":"https://github.com/dhk1349/HL-CLIP","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":4,"n_samples":7,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"R^2-Tuning","metrics":{"Hit@1":"64.20","mAP":"40.75"},"uses_additional_data":false,"paper_date":"2024-03-31","paper":"/paper/r-2-tuning-efficient-image-to-video-transfer-1","paper_url":"https://arxiv.org/abs/2404.00801v2","paper_title":"$R^2$-Tuning: Efficient Image-to-Video Transfer Learning for Video Temporal Grounding","code":"https://github.com/yeliudev/R2-Tuning","n_code_links":1,"syntology":null},{"rank_in_archive_order":7,"model":"CG-DETR (w/ PT)","metrics":{"Hit@1":"66.60","mAP":"40.71"},"uses_additional_data":true,"paper_date":"2023-11-15","paper":"/paper/correlation-guided-query-dependency","paper_url":"https://arxiv.org/abs/2311.08835v4","paper_title":"Correlation-Guided Query-Dependency Calibration for Video Temporal Grounding","code":"https://github.com/wjun0830/qd-detr","n_code_links":2,"syntology":{"n_ran":2,"n_unverified":9,"n_samples":11,"n_pointer_only_licence":11}},{"rank_in_archive_order":8,"model":"NumPro","metrics":{"Hit@1":"70.71","mAP":"40.54"},"uses_additional_data":false,"paper_date":"2024-11-15","paper":"/paper/number-it-temporal-grounding-videos-like","paper_url":"https://arxiv.org/abs/2411.10332v2","paper_title":"Number it: Temporal Grounding Videos like Flipping Manga","code":"https://github.com/yongliang-wu/numpro","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":10,"n_samples":11,"n_pointer_only_licence":0}},{"rank_in_archive_order":9,"model":"UniVTG (w/ PT)","metrics":{"Hit@1":"66.28","mAP":"40.54"},"uses_additional_data":true,"paper_date":"2023-07-31","paper":"/paper/univtg-towards-unified-video-language","paper_url":"https://arxiv.org/abs/2307.16715v2","paper_title":"UniVTG: Towards Unified Video-Language Temporal Grounding","code":"https://github.com/showlab/univtg","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":6,"n_samples":16,"n_pointer_only_licence":0}},{"rank_in_archive_order":10,"model":"CG-DETR","metrics":{"Hit@1":"66.21","mAP":"40.33"},"uses_additional_data":false,"paper_date":"2023-11-15","paper":"/paper/correlation-guided-query-dependency","paper_url":"https://arxiv.org/abs/2311.08835v4","paper_title":"Correlation-Guided Query-Dependency Calibration for Video Temporal Grounding","code":"https://github.com/wjun0830/qd-detr","n_code_links":2,"syntology":{"n_ran":2,"n_unverified":9,"n_samples":11,"n_pointer_only_licence":11}},{"rank_in_archive_order":11,"model":"LLMEPET","metrics":{"Hit@1":"65.69","mAP":"40.33"},"uses_additional_data":false,"paper_date":"2024-07-21","paper":"/paper/prior-knowledge-integration-via-llm-encoding","paper_url":"https://arxiv.org/abs/2407.15051v3","paper_title":"Prior Knowledge Integration via LLM Encoding and Pseudo Event Regulation for Video Moment Retrieval","code":"https://github.com/fletcherjiang/llmepet","n_code_links":1,"syntology":{"n_ran":5,"n_unverified":6,"n_samples":11,"n_pointer_only_licence":0}},{"rank_in_archive_order":12,"model":"UMT (w. PT)","metrics":{"mAP":"39.12"},"uses_additional_data":false,"paper_date":"2022-03-23","paper":"/paper/umt-unified-multi-modal-transformers-for","paper_url":"https://arxiv.org/abs/2203.12745v2","paper_title":"UMT: Unified Multi-modal Transformers for Joint Video Moment Retrieval and Highlight Detection","code":"https://github.com/tencentarc/umt","n_code_links":3,"syntology":null},{"rank_in_archive_order":13,"model":"QD-DETR","metrics":{"Hit@1":"62.87","mAP":"39.04"},"uses_additional_data":false,"paper_date":"2023-03-24","paper":"/paper/query-dependent-video-representation-for","paper_url":"https://arxiv.org/abs/2303.13874v1","paper_title":"Query-Dependent Video Representation for Moment Retrieval and Highlight Detection","code":"https://github.com/wjun0830/qd-detr","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":1}},{"rank_in_archive_order":14,"model":"QD-DETR (only Video)","metrics":{"Hit@1":"62.40","mAP":"38.94"},"uses_additional_data":false,"paper_date":"2023-03-24","paper":"/paper/query-dependent-video-representation-for","paper_url":"https://arxiv.org/abs/2303.13874v1","paper_title":"Query-Dependent Video Representation for Moment Retrieval and Highlight Detection","code":"https://github.com/wjun0830/qd-detr","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":1}},{"rank_in_archive_order":15,"model":"QD-DETR (w/ PT)","metrics":{"Hit@1":"62.27","mAP":"38.52"},"uses_additional_data":false,"paper_date":"2023-03-24","paper":"/paper/query-dependent-video-representation-for","paper_url":"https://arxiv.org/abs/2303.13874v1","paper_title":"Query-Dependent Video Representation for Moment Retrieval and Highlight Detection","code":"https://github.com/wjun0830/qd-detr","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":1}},{"rank_in_archive_order":16,"model":"UniVTG","metrics":{"Hit@1":"60.96","mAP":"38.20"},"uses_additional_data":false,"paper_date":"2023-07-31","paper":"/paper/univtg-towards-unified-video-language","paper_url":"https://arxiv.org/abs/2307.16715v2","paper_title":"UniVTG: Towards Unified Video-Language Temporal Grounding","code":"https://github.com/showlab/univtg","n_code_links":1,"syntology":{"n_ran":10,"n_unverified":6,"n_samples":16,"n_pointer_only_licence":0}},{"rank_in_archive_order":17,"model":"UMT","metrics":{"mAP":"38.18"},"uses_additional_data":false,"paper_date":"2022-03-23","paper":"/paper/umt-unified-multi-modal-transformers-for","paper_url":"https://arxiv.org/abs/2203.12745v2","paper_title":"UMT: Unified Multi-modal Transformers for Joint Video Moment Retrieval and Highlight Detection","code":"https://github.com/tencentarc/umt","n_code_links":3,"syntology":null},{"rank_in_archive_order":18,"model":"Moment-DETR w/ PT","metrics":{"Hit@1":"60.17","mAP":"37.43"},"uses_additional_data":false,"paper_date":"2021-07-20","paper":"/paper/qvhighlights-detecting-moments-and-highlights","paper_url":"https://arxiv.org/abs/2107.09609v2","paper_title":"QVHighlights: Detecting Moments and Highlights in Videos via Natural Language Queries","code":"https://github.com/jayleicn/moment_detr","n_code_links":4,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":1}},{"rank_in_archive_order":19,"model":"VideoChat-T (FT)","metrics":{"Hit@1":"55.3","mAP":"27.0"},"uses_additional_data":false,"paper_date":"2024-10-25","paper":"/paper/timesuite-improving-mllms-for-long-video","paper_url":"https://arxiv.org/abs/2410.19702v2","paper_title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","code":"https://github.com/OpenGVLab/TimeSuite","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":3,"n_samples":6,"n_pointer_only_licence":0}},{"rank_in_archive_order":20,"model":"VideoChat-T (ZS)","metrics":{"Hit@1":"54.1","mAP":"26.5"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":21,"model":"QD-DETR (only Video w/ PT)","metrics":{"Hit@1":"61.91"},"uses_additional_data":false,"paper_date":"2023-03-24","paper":"/paper/query-dependent-video-representation-for","paper_url":"https://arxiv.org/abs/2303.13874v1","paper_title":"Query-Dependent Video Representation for Moment Retrieval and Highlight Detection","code":"https://github.com/wjun0830/qd-detr","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":1}}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":14,"rows_with_any_sample_ran":14,"distinct_papers_with_graph_line":9,"distinct_papers_with_any_sample_ran":9,"samples_over_distinct_papers":{"n_ran":29,"n_unverified":49,"n_samples":78,"n_pointer_only_licence":26,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":44,"n_unverified":64,"n_samples":108,"n_pointer_only_licence":40,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}