{"url":"/dataset/v-coco","name":"V-COCO","full_name":"Verbs in COCO","description_markdown":"**Verbs in COCO** (**V-COCO**) is a dataset that builds off COCO for human-object interaction detection. V-COCO provides 10,346 images (2,533 for training, 2,867 for validating and 4,946 for testing) and 16,199 person instances. Each person has annotations for 29 action categories and there are no interaction labels including objects.\r\n\r\nSource: [Visual Compositional Learning for Human-Object Interaction Detection](https://arxiv.org/abs/2007.12407)\r\nImage Source: [https://www.researchgate.net/figure/Pose-estimation-and-action-recognition-results-on-the-V-COCO-Dataset-16-which-has_fig9_339477856](https://www.researchgate.net/figure/Pose-estimation-and-action-recognition-results-on-the-V-COCO-Dataset-16-which-has_fig9_339477856)","description_withheld":null,"homepage":"https://github.com/s-gupta/v-coco","introduced_date":"2015-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/visual-semantic-role-labeling","title":"Visual Semantic Role Labeling","first_author":"Saurabh Gupta","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Human-Object Interaction Detection","url":"/task/human-object-interaction-detection","datasets_with_task":"/datasets/task/human-object-interaction-detection"}],"languages":[],"variants":["V-COCO"],"data_loaders":[{"repo":"https://github.com/s-gupta/v-coco","url":"https://github.com/s-gupta/v-coco","frameworks":[]}],"num_papers_in_archive":159,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/human-object-interaction-detection-on-v-coco","task":"Human-Object Interaction Detection","dataset_variant":"V-COCO","rows":34,"metrics":["AP(S1)","AP(S2)","Time Per Frame(ms)","MAP"],"first_row_in_archive_order":{"model":"RLIPv2","paper":"/paper/rlipv2-fast-scaling-of-relational-language","metrics":{"AP(S1)":"72.1","AP(S2)":"74.1"},"code_links":[{"title":"jacobyuan7/rlipv2","url":"https://github.com/jacobyuan7/rlipv2"},{"title":"jacobyuan7/rlip","url":"https://github.com/jacobyuan7/rlip"},{"title":"jacobyuan7/ocn-hoi-benchmark","url":"https://github.com/jacobyuan7/ocn-hoi-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/rlipv2-fast-scaling-of-relational-language","title":"RLIPv2: Fast Scaling of Relational Language-Image Pre-training","date":"2023-08-18","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":30,"samples_ran":22,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/boosting-human-object-interaction-detection","title":"Boosting Human-Object Interaction Detection with Text-to-Image Diffusion Model","date":"2023-05-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/relational-context-learning-for-human-object","title":"Relational Context Learning for Human-Object Interaction Detection","date":"2023-04-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":10,"samples_unverified":2,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hoiclip-efficient-knowledge-transfer-for-hoi","title":"HOICLIP: Efficient Knowledge Transfer for HOI Detection with Vision-Language Models","date":"2023-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mining-cross-person-cues-for-body-part","title":"Mining Cross-Person Cues for Body-Part Interactiveness Learning in HOI Detection","date":"2022-07-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-skeleton-aware-graph-convolutional-network","title":"A Skeleton-aware Graph Convolutional Network for Human-Object Interaction Detection","date":"2022-07-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/exploring-structure-aware-transformer-over-1","title":"Exploring Structure-aware Transformer over Interaction Proposals for Human-Object Interaction Detection","date":"2022-06-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/consistency-learning-via-decoding-path","title":"Consistency Learning via Decoding Path Augmentation for Transformers in Human Object Interaction Detection","date":"2022-04-11","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/detecting-human-object-interactions-with-2","title":"Detecting Human-Object Interactions with Object-Guided Cross-Modal Calibrated Semantics","date":"2022-02-01","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/efficient-two-stage-detection-of-human-object","title":"Efficient Two-Stage Detection of Human-Object Interactions with a Novel Unary-Pairwise Transformer","date":"2021-12-03","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mining-the-benefits-of-two-stage-and-one","title":"Mining the Benefits of Two-stage and One-stage HOI Detection","date":"2021-08-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hotr-end-to-end-human-object-interaction","title":"HOTR: End-to-End Human-Object Interaction Detection with Transformers","date":"2021-04-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/glance-and-gaze-inferring-action-aware-points","title":"Glance and Gaze: Inferring Action-aware Points for One-Stage Human-Object Interaction Detection","date":"2021-04-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/qpic-query-based-pairwise-human-object","title":"QPIC: Query-Based Pairwise Human-Object Interaction Detection with Image-Wide Contextual Information","date":"2021-03-09","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/transferable-interactiveness-knowledge-for","title":"Transferable Interactiveness Knowledge for Human-Object Interaction Detection","date":"2021-01-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/spatio-attentive-graphs-for-human-object","title":"Spatially Conditioned Graphs for Detecting Human-Object Interactions","date":"2020-12-11","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/hoi-analysis-integrating-and-decomposing","title":"HOI Analysis: Integrating and Decomposing Human-Object Interaction","date":"2020-10-30","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/dirv-dense-interaction-region-voting-for-end","title":"DIRV: Dense Interaction Region Voting for End-to-End Human-Object Interaction Detection","date":"2020-10-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/drg-dual-relation-graph-for-human-object","title":"DRG: Dual Relation Graph for Human-Object Interaction Detection","date":"2020-08-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/polysemy-deciphering-network-for-robust-human","title":"Polysemy Deciphering Network for Robust Human-Object Interaction Detection","date":"2020-08-07","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/pose-based-modular-network-for-human-object","title":"Pose-based Modular Network for Human-Object Interaction Detection","date":"2020-08-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/detecting-human-object-interactions-with-1","title":"Detecting Human-Object Interactions with Action Co-occurrence Priors","date":"2020-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pastanet-toward-human-activity-knowledge","title":"PaStaNet: Toward Human Activity Knowledge Engine","date":"2020-04-02","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/vsgnet-spatial-attention-network-for","title":"VSGNet: Spatial Attention Network for Detecting Human Object Interactions Using Graph Convolutions","date":"2020-03-11","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ppdm-parallel-point-detection-and-matching","title":"PPDM: Parallel Point Detection and Matching for Real-time Human-Object Interaction Detection","date":"2019-12-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/transferable-interactiveness-prior-for-human","title":"Transferable Interactiveness Knowledge for Human-Object Interaction Detection","date":"2018-11-20","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ican-instance-centric-attention-network-for","title":"iCAN: Instance-Centric Attention Network for Human-Object Interaction Detection","date":"2018-08-30","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-human-object-interactions-by-graph","title":"Learning Human-Object Interactions by Graph Parsing Neural Networks","date":"2018-08-23","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":15,"samples_harvested":96,"samples_ran":50,"samples_unverified":46,"pointer_only_for_licence":17,"papers_with_no_sample_that_ran":5,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}