{"url":"/task/spatio-temporal-video-grounding","name":"Spatio-Temporal Video Grounding","slug":"spatio-temporal-video-grounding","description_markdown":"Spatio-temporal video grounding is a computer vision and natural language processing (NLP) task that involves linking textual descriptions to specific spatio-temporal regions or moments in a video. In other words, it aims to determine which parts of a video correspond to a given textual query or description. This task is essential for various applications, including video summarization, content-based video retrieval, video captioning, and more.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":22,"papers_with_code":10,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/spatio-temporal-video-grounding-on-hc-stvg2","slug":"spatio-temporal-video-grounding-on-hc-stvg2","dataset":"HC-STVG2","dataset_url":"/dataset/hc-stvg2","rows_in_archive":4,"metrics":["Val m_vIoU","Val vIoU@0.3","Val vIoU@0.5"],"first_row_in_archive_order":{"model":"TA-STVG","paper_title":"Knowing Your Target: Target-Aware Transformer Makes Better Spatio-Temporal Video Grounding","paper_url":"/paper/knowing-your-target-target-aware-transformer","paper_date":"2025-02-16","arxiv_id":"2502.11168","code_links":[{"title":"HengLan/TA-STVG","url":"https://github.com/HengLan/TA-STVG"}],"syntology":null}},{"leaderboard":"/sota/spatio-temporal-video-grounding-on-hc-stvg1","slug":"spatio-temporal-video-grounding-on-hc-stvg1","dataset":"HC-STVG1","dataset_url":"/dataset/hc-stvg1","rows_in_archive":3,"metrics":["m_vIoU","vIoU@0.3","vIoU@0.5"],"first_row_in_archive_order":{"model":"TA-STVG","paper_title":"Knowing Your Target: Target-Aware Transformer Makes Better Spatio-Temporal Video Grounding","paper_url":"/paper/knowing-your-target-target-aware-transformer","paper_date":"2025-02-16","arxiv_id":"2502.11168","code_links":[{"title":"HengLan/TA-STVG","url":"https://github.com/HengLan/TA-STVG"}],"syntology":null}},{"leaderboard":"/sota/spatio-temporal-video-grounding-on-vidstg","slug":"spatio-temporal-video-grounding-on-vidstg","dataset":"VidSTG","dataset_url":"/dataset/vidstg","rows_in_archive":3,"metrics":["Declarative m_vIoU","Declarative vIoU@0.3","Declarative vIoU@0.5","Interrogative m_vIoU","Interrogative vIoU@0.3","Interrogative vIoU@0.5"],"first_row_in_archive_order":{"model":"TA-STVG","paper_title":"Knowing Your Target: Target-Aware Transformer Makes Better Spatio-Temporal Video Grounding","paper_url":"/paper/knowing-your-target-target-aware-transformer","paper_date":"2025-02-16","arxiv_id":"2502.11168","code_links":[{"title":"HengLan/TA-STVG","url":"https://github.com/HengLan/TA-STVG"}],"syntology":null}}],"datasets":[{"url":"/dataset/vidstg","name":"VidSTG","full_name":"","num_papers_in_archive":29},{"url":"/dataset/hc-stvg2","name":"HC-STVG2","full_name":"","num_papers_in_archive":5},{"url":"/dataset/hc-stvg1","name":"HC-STVG1","full_name":"Human-centric Spatio-Temporal Video Grounding","num_papers_in_archive":4}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":10,"of":10,"tagged_in_all":22,"items":[{"url":"/paper/context-guided-spatio-temporal-video","title":"Context-Guided Spatio-Temporal Video Grounding","date":"2024-01-03","arxiv_id":"2401.01578","repositories_listed":2,"syntology":{"n":34,"n_ran":21,"n_unverified":13,"n_pointer_only":34}},{"url":"/paper/large-scale-pre-training-for-grounded-video","title":"Large-scale Pre-training for Grounded Video Caption Generation","date":"2025-03-13","arxiv_id":"2503.10781","repositories_listed":1,"syntology":null},{"url":"/paper/knowing-your-target-target-aware-transformer","title":"Knowing Your Target: Target-Aware Transformer Makes Better Spatio-Temporal Video Grounding","date":"2025-02-16","arxiv_id":"2502.11168","repositories_listed":1,"syntology":null},{"url":"/paper/pg-video-llava-pixel-grounding-large-video","title":"PG-Video-LLaVA: Pixel Grounding Large Video-Language Models","date":"2023-11-22","arxiv_id":"2311.13435","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":5}},{"url":"/paper/guided-attention-for-interpretable-motion","title":"Guided Attention for Interpretable Motion Captioning","date":"2023-10-11","arxiv_id":"2310.07324","repositories_listed":1,"syntology":null},{"url":"/paper/what-when-and-where-self-supervised-spatio","title":"What, when, and where? -- Self-Supervised Spatio-Temporal Grounding in Untrimmed Multi-Action Videos from Narrated Instructions","date":"2023-03-29","arxiv_id":"2303.16990","repositories_listed":1,"syntology":null},{"url":"/paper/embracing-consistency-a-one-stage-approach","title":"Embracing Consistency: A One-Stage Approach for Spatio-Temporal Video Grounding","date":"2022-09-27","arxiv_id":"2209.13306","repositories_listed":1,"syntology":{"n":18,"n_ran":11,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/tubedetr-spatio-temporal-video-grounding-with","title":"TubeDETR: Spatio-Temporal Video Grounding with Transformers","date":"2022-03-30","arxiv_id":"2203.16434","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/human-centric-spatio-temporal-video-grounding","title":"Human-centric Spatio-Temporal Video Grounding With Visual Transformers","date":"2020-11-10","arxiv_id":"2011.05049","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/where-does-it-exist-spatio-temporal-video","title":"Where Does It Exist: Spatio-Temporal Video Grounding for Multi-Form Sentences","date":"2020-01-19","arxiv_id":"2001.06891","repositories_listed":1,"syntology":null}],"syntology_records":5,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}