{"url":"/dataset/charades-sta","name":"Charades-STA","full_name":null,"description_markdown":"Charades-STA is a new dataset built on top of Charades by adding sentence temporal annotations.\r\n\r\nSource: [TALL: Temporal Activity Localization via Language Query](/paper/tall-temporal-activity-localization-via)","description_withheld":null,"homepage":"https://github.com/jiyanggao/TALL","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/tall-temporal-activity-localization-via","title":"TALL: Temporal Activity Localization via Language Query","first_author":"Jiyang Gao","url":null},"license":{"name":"Custom","url":"https://prior.allenai.org/projects/data/charades/license.txt"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Video Retrieval","url":"/task/video-retrieval","datasets_with_task":"/datasets/task/video-retrieval"},{"name":"Video Understanding","url":"/task/video-understanding","datasets_with_task":"/datasets/task/video-understanding"},{"name":"Moment Retrieval","url":"/task/moment-retrieval","datasets_with_task":"/datasets/task/moment-retrieval"},{"name":"Partially Relevant Video Retrieval","url":"/task/partially-relevant-video-retrieval","datasets_with_task":"/datasets/task/partially-relevant-video-retrieval"},{"name":"Decision Making","url":"/task/decision-making","datasets_with_task":"/datasets/task/decision-making"},{"name":"Temporal Sentence Grounding","url":"/task/temporal-sentence-grounding","datasets_with_task":"/datasets/task/temporal-sentence-grounding"},{"name":"Temporal Localization","url":"/task/temporal-localization","datasets_with_task":"/datasets/task/temporal-localization"}],"languages":[],"variants":["Charades-STA"],"data_loaders":[{"repo":"https://github.com/jiyanggao/TALL","url":"https://github.com/jiyanggao/TALL","frameworks":["tf"]}],"num_papers_in_archive":236,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/moment-retrieval-on-charades-sta","task":"Moment Retrieval","dataset_variant":"Charades-STA","rows":25,"metrics":["R@1 IoU=0.5","R@1 IoU=0.7","R@5 IoU=0.5","R@5 IoU=0.7","R@1 IoU=0.3","mIoU"],"first_row_in_archive_order":{"model":"SG-DETR (w/ PT)","paper":"/paper/saliency-guided-detr-for-moment-retrieval-and","metrics":{"R@1 IoU=0.5":"71.10","R@1 IoU=0.7":"52.80"},"code_links":[{"title":"ai-forever/sg-detr","url":"https://github.com/ai-forever/sg-detr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/temporal-sentence-grounding-on-charades-sta","task":"Temporal Sentence Grounding","dataset_variant":"Charades-STA","rows":13,"metrics":["R1@0.7","R1@0.5","R5@0.7","R5@0.5"],"first_row_in_archive_order":{"model":"DeCafNet","paper":"/paper/decafnet-delegate-and-conquer-for-efficient","metrics":{"R1@0.5":"68.79","R1@0.7":"47.55","R5@0.5":"91.53","R5@0.7":"72.96"},"code_links":[{"title":"zijialewislu/cvpr2025-decafnet","url":"https://github.com/zijialewislu/cvpr2025-decafnet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/partially-relevant-video-retrieval-on-1","task":"Partially Relevant Video Retrieval","dataset_variant":"Charades-STA","rows":1,"metrics":["Recall@Sum"],"first_row_in_archive_order":{"model":"ms-sl","paper":"/paper/partially-relevant-video-retrieval","metrics":{"Recall@Sum":"68.4"},"code_links":[{"title":"HuiGuanLab/ms-sl","url":"https://github.com/HuiGuanLab/ms-sl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-retrieval-on-charades-sta","task":"Video Retrieval","dataset_variant":"Charades-STA","rows":1,"metrics":["text-to-video Mean Rank","text-to-video Median Rank","text-to-video R@1","text-to-video R@10","video-to-text Mean Rank","video-to-text Median Rank","video-to-text R@1","video-to-text R@10"],"first_row_in_archive_order":{"model":"PO Loss","paper":"/paper/rudder-a-cross-lingual-video-and-text","metrics":{"text-to-video Mean Rank":"162.3","text-to-video Median Rank":"77","text-to-video R@1":"3.6","text-to-video R@10":"15.9","video-to-text Mean Rank":"164.6","video-to-text Median Rank":"83","video-to-text R@1":"3.2","video-to-text R@10":"14.9"},"code_links":[{"title":"nshubham655/RUDDER","url":"https://github.com/nshubham655/RUDDER"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/decafnet-delegate-and-conquer-for-efficient","title":"DeCafNet: Delegate and Conquer for Efficient Temporal Grounding in Long Videos","date":"2025-05-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ld-detr-loop-decoder-detection-transformer","title":"LD-DETR: Loop Decoder DEtection TRansformer for Video Moment Retrieval and Highlight Detection","date":"2025-01-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/flashvtg-feature-layering-and-adaptive-score","title":"FlashVTG: Feature Layering and Adaptive Score Handling Network for Video Temporal Grounding","date":"2024-12-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":2,"samples_unverified":11,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/videolights-feature-refinement-and-cross-task","title":"VideoLights: Feature Refinement and Cross-Task Alignment Transformer for Joint Video Highlight Detection and Moment Retrieval","date":"2024-12-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/llava-mr-large-language-and-vision-assistant","title":"LLaVA-MR: Large Language-and-Vision Assistant for Video Moment Retrieval","date":"2024-11-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/timesuite-improving-mllms-for-long-video","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","date":"2024-10-25","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/saliency-guided-detr-for-moment-retrieval-and","title":"Saliency-Guided DETR for Moment Retrieval and Highlight Detection","date":"2024-10-02","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/prior-knowledge-integration-via-llm-encoding","title":"Prior Knowledge Integration via LLM Encoding and Pseudo Event Regulation for Video Moment Retrieval","date":"2024-07-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unimd-towards-unifying-moment-retrieval-and","title":"UniMD: Towards Unifying Moment Retrieval and Temporal Action Detection","date":"2024-04-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/internvideo2-scaling-video-foundation-models","title":"InternVideo2: Scaling Foundation Models for Multimodal Video Understanding","date":"2024-03-22","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/video-mamba-suite-state-space-model-as-a","title":"Video Mamba Suite: State Space Model as a Versatile Alternative for Video Understanding","date":"2024-03-14","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":8,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bam-detr-boundary-aligned-moment-detection","title":"BAM-DETR: Boundary-Aligned Moment Detection Transformer for Temporal Sentence Grounding in Videos","date":"2023-11-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":8,"samples_unverified":3,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bridging-the-gap-a-unified-video","title":"Bridging the Gap: A Unified Video Comprehension Framework for Moment Retrieval and Highlight Detection","date":"2023-11-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":9,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/adafocus-towards-end-to-end-weakly-supervised","title":"Towards Weakly Supervised End-to-end Learning for Long-video Action Recognition","date":"2023-11-28","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/correlation-guided-query-dependency","title":"Correlation-Guided Query-Dependency Calibration for Video Temporal Grounding","date":"2023-11-15","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":2,"samples_unverified":9,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unloc-a-unified-framework-for-video","title":"UnLoc: A Unified Framework for Video Localization Tasks","date":"2023-08-21","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/d3g-exploring-gaussian-prior-for-temporal","title":"D3G: Exploring Gaussian Prior for Temporal Sentence Grounding with Glance Annotation","date":"2023-08-08","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":13,"samples_unverified":4,"pointer_only_for_licence":17,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/overcoming-weak-visual-textual-alignment-for","title":"Background-aware Moment Detection for Video Moment Retrieval","date":"2023-06-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/query-dependent-video-representation-for","title":"Query-Dependent Video Representation for Moment Retrieval and Highlight Detection","date":"2023-03-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/simvtp-simple-video-text-pre-training-with","title":"SimVTP: Simple Video Text Pre-training with Masked Autoencoders","date":"2022-12-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/partially-relevant-video-retrieval","title":"Partially Relevant Video Retrieval","date":"2022-08-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/umt-unified-multi-modal-transformers-for","title":"UMT: Unified Multi-modal Transformers for Joint Video Moment Retrieval and Highlight Detection","date":"2022-03-23","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/weakly-supervised-temporal-sentence-grounding","title":"Weakly Supervised Temporal Sentence Grounding With Gaussian-Based Contrastive Proposal Learning","date":"2022-01-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/negative-sample-matters-a-renaissance-of","title":"Negative Sample Matters: A Renaissance of Metric Learning for Temporal Grounding","date":"2021-09-10","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/qvhighlights-detecting-moments-and-highlights","title":"QVHighlights: Detecting Moments and Highlights in Videos via Natural Language Queries","date":"2021-07-20","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rudder-a-cross-lingual-video-and-text","title":"Rudder: A Cross Lingual Video and Text Retrieval Dataset","date":"2021-03-09","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":10,"samples_harvested":97,"samples_ran":53,"samples_unverified":44,"pointer_only_for_licence":54,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}