{"url":"/task/ad-hoc-video-search","name":"Ad-hoc video search","slug":"ad-hoc-video-search","description_markdown":"The Ad-hoc search task ended a 3 year cycle from 2016-2018 with a goal to model the end user search use-case, who is searching (using textual sentence queries) for segments of video containing persons, objects, activities, locations, etc. and combinations of the former. While the Internet Archive (IACC.3) dataset was adopted between 2016 to 2018, starting in 2019 a new data collection based on Vimeo Creative Commons (V3C) will be adopted to support the task for at least 3 more years.\r\n\r\nGiven the test collection (V3C1 or IACC.3), master shot boundary reference, and set of Ad-hoc queries (approx. 30 queries) released by NIST, return for each query a list of at most 1000 shot IDs from the test collection ranked according to their likelihood of containing the target query.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":13,"papers_with_code":9,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":8,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/ad-hoc-video-search-on-trecvid-avs16-iacc-3","slug":"ad-hoc-video-search-on-trecvid-avs16-iacc-3","dataset":"TRECVID-AVS16 (IACC.3)","dataset_url":"/dataset/trecvid-avs16-iacc-3","rows_in_archive":4,"metrics":["infAP"],"first_row_in_archive_order":{"model":"LAFF","paper_title":"Lightweight Attentional Feature Fusion: A New Baseline for Text-to-Video Retrieval","paper_url":"/paper/lightweight-attentional-feature-fusion-for","paper_date":"2021-12-03","arxiv_id":"2112.01832","code_links":[{"title":"ruc-aimc-lab/laff","url":"https://github.com/ruc-aimc-lab/laff"}],"syntology":null}},{"leaderboard":"/sota/ad-hoc-video-search-on-trecvid-avs17-iacc-3","slug":"ad-hoc-video-search-on-trecvid-avs17-iacc-3","dataset":"TRECVID-AVS17 (IACC.3)","dataset_url":"/dataset/trecvid-avs17-iacc-3","rows_in_archive":4,"metrics":["infAP"],"first_row_in_archive_order":{"model":"LAFF","paper_title":"Lightweight Attentional Feature Fusion: A New Baseline for Text-to-Video Retrieval","paper_url":"/paper/lightweight-attentional-feature-fusion-for","paper_date":"2021-12-03","arxiv_id":"2112.01832","code_links":[{"title":"ruc-aimc-lab/laff","url":"https://github.com/ruc-aimc-lab/laff"}],"syntology":null}},{"leaderboard":"/sota/ad-hoc-video-search-on-trecvid-avs18-iacc-3","slug":"ad-hoc-video-search-on-trecvid-avs18-iacc-3","dataset":"TRECVID-AVS18 (IACC.3)","dataset_url":"/dataset/trecvid-avs18-iacc-3","rows_in_archive":4,"metrics":["infAP"],"first_row_in_archive_order":{"model":"LAFF","paper_title":"Lightweight Attentional Feature Fusion: A New Baseline for Text-to-Video Retrieval","paper_url":"/paper/lightweight-attentional-feature-fusion-for","paper_date":"2021-12-03","arxiv_id":"2112.01832","code_links":[{"title":"ruc-aimc-lab/laff","url":"https://github.com/ruc-aimc-lab/laff"}],"syntology":null}},{"leaderboard":"/sota/ad-hoc-video-search-on-trecvid-avs19-v3c1","slug":"ad-hoc-video-search-on-trecvid-avs19-v3c1","dataset":"TRECVID-AVS19 (V3C1)","dataset_url":"/dataset/trecvid-avs19-v3c1","rows_in_archive":2,"metrics":["infAP"],"first_row_in_archive_order":{"model":"LAFF","paper_title":"Lightweight Attentional Feature Fusion: A New Baseline for Text-to-Video Retrieval","paper_url":"/paper/lightweight-attentional-feature-fusion-for","paper_date":"2021-12-03","arxiv_id":"2112.01832","code_links":[{"title":"ruc-aimc-lab/laff","url":"https://github.com/ruc-aimc-lab/laff"}],"syntology":null}},{"leaderboard":"/sota/ad-hoc-video-search-on-trecvid-avs20-v3c1","slug":"ad-hoc-video-search-on-trecvid-avs20-v3c1","dataset":"TRECVID-AVS20 (V3C1)","dataset_url":"/dataset/trecvid-avs20-v3c1","rows_in_archive":1,"metrics":["infAP"],"first_row_in_archive_order":{"model":"LAFF","paper_title":"Lightweight Attentional Feature Fusion: A New Baseline for Text-to-Video Retrieval","paper_url":"/paper/lightweight-attentional-feature-fusion-for","paper_date":"2021-12-03","arxiv_id":"2112.01832","code_links":[{"title":"ruc-aimc-lab/laff","url":"https://github.com/ruc-aimc-lab/laff"}],"syntology":null}}],"datasets":[{"url":"/dataset/trecvid","name":"TRECVID","full_name":"TRECVID","num_papers_in_archive":6},{"url":"/dataset/trecvid-avs16-iacc-3","name":"TRECVID-AVS16 (IACC.3)","full_name":"","num_papers_in_archive":5},{"url":"/dataset/trecvid-avs17-iacc-3","name":"TRECVID-AVS17 (IACC.3)","full_name":"","num_papers_in_archive":5},{"url":"/dataset/trecvid-avs18-iacc-3","name":"TRECVID-AVS18 (IACC.3)","full_name":"","num_papers_in_archive":5},{"url":"/dataset/trecvid-avs19-v3c1","name":"TRECVID-AVS19 (V3C1)","full_name":"","num_papers_in_archive":3},{"url":"/dataset/iacc-3","name":"IACC.3","full_name":"Internet Archive videos (IACC.3) under Creative Commons licenses.","num_papers_in_archive":2},{"url":"/dataset/trecvid-avs20-v3c1","name":"TRECVID-AVS20 (V3C1)","full_name":"","num_papers_in_archive":1},{"url":"/dataset/trecvid-avs21-v3c1","name":"TRECVID-AVS21 (V3C1)","full_name":"","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":9,"of":9,"tagged_in_all":13,"items":[{"url":"/paper/interpretable-embedding-for-ad-hoc-video","title":"Interpretable Embedding for Ad-hoc Video Search","date":"2024-02-19","arxiv_id":"2402.11812","repositories_listed":1,"syntology":null},{"url":"/paper/an-overview-on-the-evaluated-video-retrieval","title":"An overview on the evaluated video retrieval tasks at TRECVID 2022","date":"2023-06-22","arxiv_id":"2306.13118","repositories_listed":1,"syntology":null},{"url":"/paper/un-likelihood-training-for-interpretable","title":"(Un)likelihood Training for Interpretable Embedding","date":"2022-07-01","arxiv_id":"2207.00282","repositories_listed":1,"syntology":null},{"url":"/paper/lightweight-attentional-feature-fusion-for","title":"Lightweight Attentional Feature Fusion: A New Baseline for Text-to-Video Retrieval","date":"2021-12-03","arxiv_id":"2112.01832","repositories_listed":1,"syntology":null},{"url":"/paper/trecvid-2020-a-comprehensive-campaign-for","title":"TRECVID 2020: A comprehensive campaign for evaluating video retrieval tasks across multiple application domains","date":"2021-04-27","arxiv_id":"2104.13473","repositories_listed":1,"syntology":null},{"url":"/paper/sea-sentence-encoder-assembly-for-video","title":"SEA: Sentence Encoder Assembly for Video Retrieval by Textual Queries","date":"2020-11-24","arxiv_id":"2011.12091","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-space-learning-for-language-based","title":"Dual Encoding for Video Retrieval by Text","date":"2020-09-10","arxiv_id":"2009.05381","repositories_listed":1,"syntology":null},{"url":"/paper/w2vv-fully-deep-learning-for-ad-hoc-video","title":"W2VV++: Fully Deep Learning for Ad-hoc Video Search","date":"2019-10-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dual-dense-encoding-for-zero-example-video","title":"Dual Encoding for Zero-Example Video Retrieval","date":"2018-09-17","arxiv_id":"1809.06181","repositories_listed":1,"syntology":{"n":14,"n_ran":2,"n_unverified":12,"n_pointer_only":0}}],"syntology_records":1,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}