{"url":"/dataset/coin","name":"COIN","full_name":null,"description_markdown":"The **COIN** dataset (a large-scale dataset for COmprehensive INstructional video analysis) consists of 11,827 videos related to 180 different tasks in 12 domains (e.g., vehicles, gadgets, etc.) related to our daily life. The videos are all collected from YouTube. The average length of a video is 2.36 minutes. Each video is labelled with 3.91 step segments, where each segment lasts 14.91 seconds on average. In total, the dataset contains videos of 476 hours, with 46,354 annotated segments.\r\n\r\nSource: [COIN: A Large-scale Dataset for Comprehensive Instructional Video Analysis](/paper/coin-a-large-scale-dataset-for-comprehensive)","description_withheld":null,"homepage":"https://coin-dataset.github.io/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/coin-a-large-scale-dataset-for-comprehensive","title":"COIN: A Large-scale Dataset for Comprehensive Instructional Video Analysis","first_author":"Yansong Tang","url":null},"license":{"name":"Custom","url":"https://coin-dataset.github.io/"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Action Localization","url":"/task/action-localization","datasets_with_task":"/datasets/task/action-localization"},{"name":"Temporal Action Localization","url":"/task/action-recognition","datasets_with_task":"/datasets/task/action-recognition"},{"name":"Action Segmentation","url":"/task/action-segmentation","datasets_with_task":"/datasets/task/action-segmentation"},{"name":"Video Classification","url":"/task/video-classification","datasets_with_task":"/datasets/task/video-classification"}],"languages":[],"variants":["COIN"],"data_loaders":[],"num_papers_in_archive":105,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/action-segmentation-on-coin","task":"Action Segmentation","dataset_variant":"COIN","rows":9,"metrics":["Frame accuracy"],"first_row_in_archive_order":{"model":"UnLoc-L","paper":"/paper/unloc-a-unified-framework-for-video","metrics":{"Frame accuracy":"72.8"},"code_links":[{"title":"google-research/scenic","url":"https://github.com/google-research/scenic"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/video-classification-on-coin-1","task":"Video Classification","dataset_variant":"COIN","rows":7,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"HERMES","paper":"/paper/bridging-episodes-and-semantics-a-novel","metrics":{"Accuracy (%)":"93.5"},"code_links":[{"title":"joslefaure/HERMES","url":"https://github.com/joslefaure/HERMES"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/bridging-episodes-and-semantics-a-novel","title":"HERMES: temporal-coHERent long-forM understanding with Episodes and Semantics","date":"2024-08-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":7,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":7,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-granularity-correspondence-learning-1","title":"Multi-granularity Correspondence Learning from Long-term Noisy Videos","date":"2024-01-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unloc-a-unified-framework-for-video","title":"UnLoc: A Unified Framework for Video Localization Tasks","date":"2023-08-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/selective-structured-state-spaces-for-long","title":"Selective Structured State-Spaces for Long-Form Video Understanding","date":"2023-03-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/efficient-movie-scene-detection-using-state","title":"Efficient Movie Scene Detection using State-Space Transformers","date":"2022-12-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/long-movie-clip-classification-with-state","title":"Long Movie Clip Classification with State-Space Video Models","date":"2022-04-04","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":11,"samples_unverified":7,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-to-recognize-procedural-activities","title":"Learning To Recognize Procedural Activities with Distant Supervision","date":"2022-01-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/videoclip-contrastive-pre-training-for-zero","title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","date":"2021-09-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/taco-token-aware-cascade-contrastive-learning","title":"TACo: Token-aware Cascade Contrastive Learning for Video-Text Alignment","date":"2021-08-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/vlm-task-agnostic-video-language-model-pre","title":"VLM: Task-agnostic Video-Language Model Pre-training for Video Understanding","date":"2021-05-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/actbert-learning-global-local-video-text-1","title":"ActBERT: Learning Global-Local Video-Text Representations","date":"2020-11-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/univilm-a-unified-video-and-language-pre","title":"UniVL: A Unified Video and Language Pre-Training Model for Multimodal Understanding and Generation","date":"2020-02-15","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-learning-of-visual-representations","title":"End-to-End Learning of Visual Representations from Uncurated Instructional Videos","date":"2019-12-13","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/temporal-segment-networks-for-action","title":"Temporal Segment Networks for Action Recognition in Videos","date":"2017-05-08","rows_on_this_dataset":1,"code_links":11,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":51,"samples_ran":34,"samples_unverified":17,"pointer_only_for_licence":3,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}