{"url":"/task/temporal-localization","name":"Temporal Localization","slug":"temporal-localization","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":153,"papers_with_code":76,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":2,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/charades-sta","name":"Charades-STA","full_name":"","num_papers_in_archive":236},{"url":"/dataset/ave","name":"AVE","full_name":"Audio-Visual Event Localization","num_papers_in_archive":103},{"url":"/dataset/vidstg","name":"VidSTG","full_name":"","num_papers_in_archive":29},{"url":"/dataset/tumtraffic-videoqa","name":"TUMTraffic-VideoQA","full_name":"","num_papers_in_archive":1}],"subtasks":[{"url":"/task/language-based-temporal-localization","name":"Language-Based Temporal Localization"},{"url":"/task/temporal-defect-localization","name":"Temporal Defect Localization"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":76,"tagged_in_all":153,"items":[{"url":"/paper/tall-temporal-activity-localization-via","title":"TALL: Temporal Activity Localization via Language Query","date":"2017-05-05","arxiv_id":"1705.02101","repositories_listed":12,"syntology":null},{"url":"/paper/mac-mining-activity-concepts-for-language","title":"MAC: Mining Activity Concepts for Language-based Temporal Localization","date":"2018-11-21","arxiv_id":"1811.08925","repositories_listed":3,"syntology":null},{"url":"/paper/weakly-supervised-action-localization-by","title":"Weakly Supervised Action Localization by Sparse Temporal Pooling Network","date":"2017-12-14","arxiv_id":"1712.05080","repositories_listed":3,"syntology":null},{"url":"/paper/timechat-a-time-sensitive-multimodal-large","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","date":"2023-12-04","arxiv_id":"2312.02051","repositories_listed":2,"syntology":{"n":11,"n_ran":7,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/egocentric-video-language-pretraining","title":"Egocentric Video-Language Pretraining","date":"2022-06-03","arxiv_id":"2206.01670","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/temporal-localization-of-moments-in-video","title":"Finding Moments in Video Collections Using Natural Language","date":"2019-07-30","arxiv_id":"1907.12763","repositories_listed":2,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/190513313","title":"Technical Report of the Video Event Reconstruction and Analysis (VERA) System -- Shooter Localization, Models, Interface, and Beyond","date":"2019-05-26","arxiv_id":"1905.13313","repositories_listed":2,"syntology":null},{"url":"/paper/audio-visual-event-localization-in","title":"Audio-Visual Event Localization in Unconstrained Videos","date":"2018-03-23","arxiv_id":"1803.08842","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/hacs-human-action-clips-and-segments-dataset","title":"HACS: Human Action Clips and Segments Dataset for Recognition and Temporal Localization","date":"2017-12-26","arxiv_id":"1712.09374","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/asynchronous-temporal-fields-for-action","title":"Asynchronous Temporal Fields for Action Recognition","date":"2016-12-19","arxiv_id":"1612.06371","repositories_listed":2,"syntology":null},{"url":"/paper/videomolmo-spatio-temporal-grounding-meets","title":"VideoMolmo: Spatio-Temporal Grounding Meets Pointing","date":"2025-06-05","arxiv_id":"2506.05336","repositories_listed":1,"syntology":null},{"url":"/paper/distime-distribution-based-time","title":"DisTime: Distribution-based Time Representation for Video Large Language Models","date":"2025-05-30","arxiv_id":"2505.24329","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/transforming-faces-into-video-stories","title":"Transforming faces into video stories -- VideoFace2.0","date":"2025-05-04","arxiv_id":"2505.02060","repositories_listed":1,"syntology":null},{"url":"/paper/minerva-evaluating-complex-video-reasoning","title":"MINERVA: Evaluating Complex Video Reasoning","date":"2025-05-01","arxiv_id":"2505.00681","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-and-multimodal-data-for-daily","title":"Hierarchical and Multimodal Data for Daily Activity Understanding","date":"2025-04-24","arxiv_id":"2504.17696","repositories_listed":1,"syntology":null},{"url":"/paper/atars-an-aerial-traffic-atomic-activity","title":"ATARS: An Aerial Traffic Atomic Activity Recognition and Temporal Segmentation Dataset","date":"2025-03-24","arxiv_id":"2503.18553","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-to-the-unknown-training-free-audio","title":"Adapting to the Unknown: Training-Free Audio-Visual Event Perception with Dynamic Thresholds","date":"2025-03-17","arxiv_id":"2503.13693","repositories_listed":1,"syntology":null},{"url":"/paper/videomind-a-chain-of-lora-agent-for-long","title":"VideoMind: A Chain-of-LoRA Agent for Long Video Reasoning","date":"2025-03-17","arxiv_id":"2503.13444","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/crab-a-unified-audio-visual-scene-1","title":"Crab: A Unified Audio-Visual Scene Understanding Model with Explicit Cooperation","date":"2025-03-17","arxiv_id":"2503.13068","repositories_listed":1,"syntology":null},{"url":"/paper/timeloc-a-unified-end-to-end-framework-for","title":"TimeLoc: A Unified End-to-End Framework for Precise Timestamp Localization in Long Videos","date":"2025-03-09","arxiv_id":"2503.06526","repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-multiple-instance-learning-1","title":"Weakly Supervised Multiple Instance Learning for Whale Call Detection and Temporal Localization in Long-Duration Passive Acoustic Monitoring","date":"2025-02-28","arxiv_id":"2502.20838","repositories_listed":1,"syntology":null},{"url":"/paper/knowing-your-target-target-aware-transformer","title":"Knowing Your Target: Target-Aware Transformer Makes Better Spatio-Temporal Video Grounding","date":"2025-02-16","arxiv_id":"2502.11168","repositories_listed":1,"syntology":null},{"url":"/paper/llava-st-a-multimodal-large-language-model","title":"LLaVA-ST: A Multimodal Large Language Model for Fine-Grained Spatial-Temporal Understanding","date":"2025-01-14","arxiv_id":"2501.08282","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/do-current-video-llms-have-strong-ocr","title":"Do Current Video LLMs Have Strong OCR Abilities? A Preliminary Study","date":"2024-12-29","arxiv_id":"2412.20613","repositories_listed":1,"syntology":null},{"url":"/paper/timerefine-temporal-grounding-with-time","title":"TimeRefine: Temporal Grounding with Time Refining Video LLM","date":"2024-12-12","arxiv_id":"2412.09601","repositories_listed":1,"syntology":null},{"url":"/paper/timemarker-a-versatile-video-llm-for-long-and","title":"TimeMarker: A Versatile Video-LLM for Long and Short Video Understanding with Superior Temporal Localization Ability","date":"2024-11-27","arxiv_id":"2411.18211","repositories_listed":1,"syntology":null},{"url":"/paper/number-it-temporal-grounding-videos-like","title":"Number it: Temporal Grounding Videos like Flipping Manga","date":"2024-11-15","arxiv_id":"2411.10332","repositories_listed":1,"syntology":{"n":11,"n_ran":1,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/training-free-video-temporal-grounding-using","title":"Training-free Video Temporal Grounding using Large-scale Pre-trained Models","date":"2024-08-29","arxiv_id":"2408.16219","repositories_listed":1,"syntology":null},{"url":"/paper/meerkat-audio-visual-large-language-model-for","title":"Meerkat: Audio-Visual Large Language Model for Grounding in Space and Time","date":"2024-07-01","arxiv_id":"2407.01851","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/ophnet-a-large-scale-video-benchmark-for","title":"OphNet: A Large-Scale Video Benchmark for Ophthalmic Surgical Workflow Understanding","date":"2024-06-11","arxiv_id":"2406.07471","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_unverified":6,"n_pointer_only":0}}],"syntology_records":11,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}