{"url":"/dataset/crosstask","name":"CrossTask","full_name":"CrossTask","description_markdown":"**CrossTask** dataset contains instructional videos, collected for 83 different tasks. For each task an ordered list of steps with manual descriptions is provided. The dataset is divided in two parts: 18 primary and 65 related tasks. Videos for the primary tasks are collected manually and provided with annotations for temporal step boundaries. Videos for the related tasks are collected automatically and don't have annotations.\r\n\r\nSource: [CrossTask](https://github.com/DmZhukov/CrossTask)\r\nImage Source: [https://arxiv.org/pdf/1903.08225v2.pdf](https://arxiv.org/pdf/1903.08225v2.pdf)","description_withheld":null,"homepage":"https://github.com/DmZhukov/CrossTask","introduced_date":"2019-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/cross-task-weakly-supervised-learning-from","title":"Cross-task weakly supervised learning from instructional videos","first_author":"Dimitri Zhukov","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Temporal Action Localization","url":"/task/action-recognition","datasets_with_task":"/datasets/task/action-recognition"}],"languages":[],"variants":["CrossTask"],"data_loaders":[{"repo":"https://github.com/DmZhukov/CrossTask","url":"https://github.com/DmZhukov/CrossTask","frameworks":["pytorch"]}],"num_papers_in_archive":54,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/temporal-action-localization-on-crosstask","task":"Temporal Action Localization","dataset_variant":"CrossTask","rows":7,"metrics":["Recall"],"first_row_in_archive_order":{"model":"VideoCLIP","paper":"/paper/videoclip-contrastive-pre-training-for-zero","metrics":{"Recall":"47.3"},"code_links":[{"title":"facebookresearch/fairseq","url":"https://github.com/facebookresearch/fairseq"},{"title":"pytorch/fairseq","url":"https://github.com/pytorch/fairseq"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/videoclip-contrastive-pre-training-for-zero","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","date":"2021-09-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/taco-token-aware-cascade-contrastive-learning","title":"TACo: Token-aware Cascade Contrastive Learning for Video-Text Alignment","date":"2021-08-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/vlm-task-agnostic-video-language-model-pre","title":"VLM: Task-agnostic Video-Language Model Pre-training for Video Understanding","date":"2021-05-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/howto100m-learning-a-text-video-embedding-by","title":"HowTo100M: Learning a Text-Video Embedding by Watching Hundred Million Narrated Video Clips","date":"2019-06-07","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-task-weakly-supervised-learning-from","title":"Cross-task weakly supervised learning from instructional videos","date":"2019-03-19","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/unsupervised-learning-from-narrated","title":"Unsupervised Learning from Narrated Instruction Videos","date":"2015-06-30","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}