{"url":"/dataset/cater","name":"CATER","full_name":null,"description_markdown":"Rendered synthetically using a library of standard 3D objects, and tests the ability to recognize compositions of object movements that require long-term reasoning.\r\n\r\nSource: [CATER: A diagnostic dataset for Compositional Actions and TEmporal Reasoning](/paper/cater-a-diagnostic-dataset-for-compositional)\r\n\r\nImage Source: [CATER](https://rohitgirdhar.github.io/CATER/)","description_withheld":null,"homepage":"https://rohitgirdhar.github.io/CATER/","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/cater-a-diagnostic-dataset-for-compositional","title":"CATER: A diagnostic dataset for Compositional Actions and TEmporal Reasoning","first_author":"Rohit Girdhar","url":null},"license":{"name":"Apache License 2.0","url":"https://github.com/rohitgirdhar/CATER/blob/master/LICENSE"},"modalities":[{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Action Recognition","url":"/task/action-recognition-in-videos","datasets_with_task":"/datasets/task/action-recognition-in-videos"},{"name":"Visual Reasoning","url":"/task/visual-reasoning","datasets_with_task":"/datasets/task/visual-reasoning"},{"name":"Self-Supervised Learning","url":"/task/self-supervised-learning","datasets_with_task":"/datasets/task/self-supervised-learning"},{"name":"Video Object Tracking","url":"/task/video-object-tracking","datasets_with_task":"/datasets/task/video-object-tracking"},{"name":"Atomic action recognition","url":"/task/atomic-action-recognition","datasets_with_task":"/datasets/task/atomic-action-recognition"},{"name":"Composite action recognition","url":"/task/composite-action-recognition","datasets_with_task":"/datasets/task/composite-action-recognition"}],"languages":[],"variants":["CATER"],"data_loaders":[],"num_papers_in_archive":51,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/video-object-tracking-on-cater","task":"Video Object Tracking","dataset_variant":"CATER","rows":7,"metrics":["Top 1 Accuracy","L1","Top 5 Accuracy"],"first_row_in_archive_order":{"model":"Loci","paper":"/paper/learning-what-and-where-unsupervised","metrics":{"L1":"0.14","Top 1 Accuracy":"90.7","Top 5 Accuracy":"98.5"},"code_links":[{"title":"CognitiveModeling/Loci","url":"https://github.com/CognitiveModeling/Loci"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/atomic-action-recognition-on-cater","task":"Atomic action recognition","dataset_variant":"CATER","rows":4,"metrics":["Average-mAP"],"first_row_in_archive_order":{"model":"SCI3D","paper":"/paper/deep-set-conditioned-latent-representations-1","metrics":{"Average-mAP":"96.77"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/composite-action-recognition-on-cater","task":"Composite action recognition","dataset_variant":"CATER","rows":4,"metrics":["Average-mAP"],"first_row_in_archive_order":{"model":"Single stream SCI3D","paper":"/paper/deep-set-conditioned-latent-representations-1","metrics":{"Average-mAP":"69.76"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/deep-set-conditioned-latent-representations-1","title":"Deep set conditioned latent representations for action recognition","date":"2022-12-21","rows_on_this_dataset":8,"code_links":0,"syntology":null},{"paper":"/paper/learning-what-and-where-unsupervised","title":"Learning What and Where: Disentangling Location and Identity Tracking Without Supervision","date":"2022-05-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tfcnet-temporal-fully-connected-networks-for","title":"TFCNet: Temporal Fully Connected Networks for Static Unbiased Temporal Reasoning","date":"2022-03-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/inferno-inferring-object-centric-3d-scene","title":"INFERNO: Inferring Object-Centric 3D Scene Representations without Supervision","date":"2021-09-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hopper-multi-hop-transformer-for-1","title":"Hopper: Multi-hop Transformer for Spatiotemporal Reasoning","date":"2021-03-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/object-based-attention-for-spatio-temporal","title":"Attention over learned object embeddings enables complex visual reasoning","date":"2020-12-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-object-permanence-from-video","title":"Learning Object Permanence from Video","date":"2020-03-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/quo-vadis-action-recognition-a-new-model-and","title":"Quo Vadis, Action Recognition? A New Model and the Kinetics Dataset","date":"2017-05-22","rows_on_this_dataset":1,"code_links":34,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":28,"samples_ran":16,"samples_unverified":12,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":2,"samples_harvested":32,"samples_ran":16,"samples_unverified":16,"pointer_only_for_licence":7,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}