{"url":"/dataset/hico-det","name":"HICO-DET","full_name":"HICO-DET","description_markdown":"**HICO-DET** is a dataset for detecting human-object interactions (HOI) in images. It contains 47,776 images (38,118 in train set and 9,658 in test set), 600 HOI categories constructed by 80 object categories and 117 verb classes. HICO-DET provides more than 150k annotated human-object pairs. V-COCO provides 10,346 images (2,533 for training, 2,867 for validating and 4,946 for testing) and 16,199 person instances. Each person has annotations for 29 action categories and there are no interaction labels including objects.\r\n\r\nSource: [Visual Compositional Learning for Human-Object Interaction Detection](https://arxiv.org/abs/2007.12407)\r\nImage Source: [http://www-personal.umich.edu/~ywchao/hico/](http://www-personal.umich.edu/~ywchao/hico/)","description_withheld":null,"homepage":"https://umich-ywchao-hico.github.io/","introduced_date":"2018-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/learning-to-detect-human-object-interactions","title":"Learning to Detect Human-Object Interactions","first_author":"Yu-Wei Chao","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Weakly Supervised Object Detection","url":"/task/weakly-supervised-object-detection","datasets_with_task":"/datasets/task/weakly-supervised-object-detection"},{"name":"Human-Object Interaction Detection","url":"/task/human-object-interaction-detection","datasets_with_task":"/datasets/task/human-object-interaction-detection"},{"name":"Zero-Shot Human-Object Interaction Detection","url":"/task/zero-shot-human-object-interaction-detection","datasets_with_task":"/datasets/task/zero-shot-human-object-interaction-detection"},{"name":"Affordance Recognition","url":"/task/affordance-recognition","datasets_with_task":"/datasets/task/affordance-recognition"},{"name":"Human-Object Interaction Concept Discovery","url":"/task/human-object-interaction-concept-discovery","datasets_with_task":"/datasets/task/human-object-interaction-concept-discovery"}],"languages":[],"variants":["HICO-DET"],"data_loaders":[],"num_papers_in_archive":180,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/human-object-interaction-detection-on-hico","task":"Human-Object Interaction Detection","dataset_variant":"HICO-DET","rows":55,"metrics":["mAP","Time Per Frame (ms)","Detection: Full (mAP@0.5)","Detection: Non-Rare (mAP@0.5)","Detection: Rare (mAP@0.5)"],"first_row_in_archive_order":{"model":"Ours (PViC+)","paper":"/paper/dynamic-scene-understanding-from-vision","metrics":{"mAP":"46.49"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/affordance-recognition-on-hico-det","task":"Affordance Recognition","dataset_variant":"HICO-DET","rows":4,"metrics":["COCO-Val2017","Object365","HICO","Novel classes"],"first_row_in_archive_order":{"model":"SCL","paper":"/paper/discovering-human-object-interaction-concepts","metrics":{"COCO-Val2017":"72.08","HICO":"82.47","Novel classes":"18.55","Object365":"57.53"},"code_links":[{"title":"zhihou7/HOI-CL","url":"https://github.com/zhihou7/HOI-CL"},{"title":"zhihou7/scl","url":"https://github.com/zhihou7/scl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/weakly-supervised-object-detection-on-hico","task":"Weakly Supervised Object Detection","dataset_variant":"HICO-DET","rows":4,"metrics":["MAP"],"first_row_in_archive_order":{"model":"Spatial Prior","paper":"/paper/activity-driven-weakly-supervised-object","metrics":{"MAP":"5.39"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/human-object-interaction-concept-discovery-on","task":"Human-Object Interaction Concept Discovery","dataset_variant":"HICO-DET","rows":3,"metrics":["Unknown (AP)"],"first_row_in_archive_order":{"model":"SCL","paper":"/paper/discovering-human-object-interaction-concepts","metrics":{"Unknown (AP)":"33.58"},"code_links":[{"title":"zhihou7/HOI-CL","url":"https://github.com/zhihou7/HOI-CL"},{"title":"zhihou7/scl","url":"https://github.com/zhihou7/scl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/dynamic-scene-understanding-from-vision","title":"Dynamic Scene Understanding from Vision-Language Representations","date":"2025-01-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/unseen-no-more-unlocking-the-potential-of","title":"Unseen No More: Unlocking the Potential of CLIP for Generative Zero-shot HOI Detection","date":"2024-08-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rlipv2-fast-scaling-of-relational-language","title":"RLIPv2: Fast Scaling of Relational Language-Image Pre-training","date":"2023-08-18","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":30,"samples_ran":22,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploring-predicate-visual-context-in","title":"Exploring Predicate Visual Context in Detecting Human-Object Interactions","date":"2023-08-11","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":3,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/focusing-on-what-to-decode-and-what-to-train","title":"Focusing on what to decode and what to train: SOV Decoding with Specific Target Guided DeNoising and Vision Language Advisor","date":"2023-07-05","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/boosting-human-object-interaction-detection","title":"Boosting Human-Object Interaction Detection with Text-to-Image Diffusion Model","date":"2023-05-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/viplo-vision-transformer-based-pose","title":"ViPLO: Vision Transformer based Pose-Conditioned Self-Loop Graph for Human-Object Interaction Detection","date":"2023-04-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/relational-context-learning-for-human-object","title":"Relational Context Learning for Human-Object Interaction Detection","date":"2023-04-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":10,"samples_unverified":2,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hoiclip-efficient-knowledge-transfer-for-hoi","title":"HOICLIP: Efficient Knowledge Transfer for HOI Detection with Vision-Language Models","date":"2023-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/category-query-learning-for-human-object","title":"Category Query Learning for Human-Object Interaction Classification","date":"2023-03-24","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ernet-efficient-and-reliable-human-object","title":"ERNet: Efficient and Reliable Human-Object Interaction Detection","date":"2023-01-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/fgahoi-fine-grained-anchors-for-human-object","title":"FGAHOI: Fine-Grained Anchors for Human-Object Interaction Detection","date":"2023-01-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rlip-relational-language-image-pre-training","title":"RLIP: Relational Language-Image Pre-training for Human-Object Interaction Detection","date":"2022-09-05","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mining-cross-person-cues-for-body-part","title":"Mining Cross-Person Cues for Body-Part Interactiveness Learning in HOI Detection","date":"2022-07-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploring-structure-aware-transformer-over-1","title":"Exploring Structure-aware Transformer over Interaction Proposals for Human-Object Interaction Detection","date":"2022-06-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/consistency-learning-via-decoding-path","title":"Consistency Learning via Decoding Path Augmentation for Transformers in Human Object Interaction Detection","date":"2022-04-11","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/discovering-human-object-interaction-concepts","title":"Discovering Human-Object Interaction Concepts via Self-Compositional Learning","date":"2022-03-27","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gen-vlkt-simplify-association-and-enhance","title":"GEN-VLKT: Simplify Association and Enhance Interaction Understanding for HOI Detection","date":"2022-03-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/detecting-human-object-interactions-with-2","title":"Detecting Human-Object Interactions with Object-Guided Cross-Modal Calibrated Semantics","date":"2022-02-01","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/qahoi-query-based-anchors-for-human-object","title":"QAHOI: Query-Based Anchors for Human-Object Interaction Detection","date":"2021-12-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":10,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/decoupling-object-detection-from-human-object-1","title":"The Overlooked Classifier in Human-Object Interaction Recognition","date":"2021-12-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/efficient-two-stage-detection-of-human-object","title":"Efficient Two-Stage Detection of Human-Object Interactions with a Novel Unary-Pairwise Transformer","date":"2021-12-03","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/acp-action-co-occurrence-priors-for-human","title":"ACP++: Action Co-occurrence Priors for Human-Object Interaction Detection","date":"2021-09-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mining-the-benefits-of-two-stage-and-one","title":"Mining the Benefits of Two-stage and One-stage HOI Detection","date":"2021-08-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hotr-end-to-end-human-object-interaction","title":"HOTR: End-to-End Human-Object Interaction Detection with Transformers","date":"2021-04-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/affordance-transfer-learning-for-human-object","title":"Affordance Transfer Learning for Human-Object Interaction Detection","date":"2021-04-07","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/detecting-human-object-interaction-via","title":"Detecting Human-Object Interaction via Fabricated Compositional Learning","date":"2021-03-15","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/reformulating-hoi-detection-as-adaptive-set","title":"Reformulating HOI Detection as Adaptive Set Prediction","date":"2021-03-10","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/qpic-query-based-pairwise-human-object","title":"QPIC: Query-Based Pairwise Human-Object Interaction Detection with Image-Wide Contextual Information","date":"2021-03-09","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-human-object-interaction-detection","title":"End-to-End Human Object Interaction Detection with HOI Transformer","date":"2021-03-08","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/transferable-interactiveness-knowledge-for","title":"Transferable Interactiveness Knowledge for Human-Object Interaction Detection","date":"2021-01-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/spatio-attentive-graphs-for-human-object","title":"Spatially Conditioned Graphs for Detecting Human-Object Interactions","date":"2020-12-11","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/hoi-analysis-integrating-and-decomposing","title":"HOI Analysis: Integrating and Decomposing Human-Object Interaction","date":"2020-10-30","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/dirv-dense-interaction-region-voting-for-end","title":"DIRV: Dense Interaction Region Voting for End-to-End Human-Object Interaction Detection","date":"2020-10-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/drg-dual-relation-graph-for-human-object","title":"DRG: Dual Relation Graph for Human-Object Interaction Detection","date":"2020-08-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/consnet-learning-consistency-graph-for-zero","title":"ConsNet: Learning Consistency Graph for Zero-Shot Human-Object Interaction Detection","date":"2020-08-14","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/polysemy-deciphering-network-for-robust-human","title":"Polysemy Deciphering Network for Robust Human-Object Interaction Detection","date":"2020-08-07","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/pose-based-modular-network-for-human-object","title":"Pose-based Modular Network for Human-Object Interaction Detection","date":"2020-08-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/detecting-human-object-interactions-with-1","title":"Detecting Human-Object Interactions with Action Co-occurrence Priors","date":"2020-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/visual-compositional-learning-for-human","title":"Visual Compositional Learning for Human-Object Interaction Detection","date":"2020-07-24","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/detailed-2d-3d-joint-representation-for-human","title":"Detailed 2D-3D Joint Representation for Human-Object Interaction","date":"2020-04-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pastanet-toward-human-activity-knowledge","title":"PaStaNet: Toward Human Activity Knowledge Engine","date":"2020-04-02","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/vsgnet-spatial-attention-network-for","title":"VSGNet: Spatial Attention Network for Detecting Human Object Interactions Using Graph Convolutions","date":"2020-03-11","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ppdm-parallel-point-detection-and-matching","title":"PPDM: Parallel Point Detection and Matching for Real-time Human-Object Interaction Detection","date":"2019-12-30","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/activity-driven-weakly-supervised-object","title":"Activity Driven Weakly Supervised Object Detection","date":"2019-04-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/transferable-interactiveness-prior-for-human","title":"Transferable Interactiveness Knowledge for Human-Object Interaction Detection","date":"2018-11-20","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ican-instance-centric-attention-network-for","title":"iCAN: Instance-Centric Attention Network for Human-Object Interaction Detection","date":"2018-08-30","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-human-object-interactions-by-graph","title":"Learning Human-Object Interactions by Graph Parsing Neural Networks","date":"2018-08-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pcl-proposal-cluster-learning-for-weakly","title":"PCL: Proposal Cluster Learning for Weakly Supervised Object Detection","date":"2018-07-09","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":1,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/detecting-and-recognizing-human-object","title":"Detecting and Recognizing Human-Object Interactions","date":"2017-04-24","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/weakly-supervised-deep-detection-networks","title":"Weakly Supervised Deep Detection Networks","date":"2015-11-09","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/contextual-action-recognition-with-rcnn","title":"Contextual Action Recognition with R*CNN","date":"2015-05-05","rows_on_this_dataset":1,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":27,"samples_harvested":144,"samples_ran":65,"samples_unverified":79,"pointer_only_for_licence":19,"papers_with_no_sample_that_ran":11,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}