{"url":"/task/human-object-interaction-detection","name":"Human-Object Interaction Detection","slug":"human-object-interaction-detection","description_markdown":"Human-Object Interaction (HOI) detection is a task of identifying \"a set of interactions\" in an image, which involves the i) localization of the subject (i.e., humans) and target (i.e., objects) of interaction, and ii) the classification of the interaction labels.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":449,"papers_with_code":173,"benchmarks":6,"benchmark_tables_in_archive":6,"benchmark_tables_shown":6,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":23,"subtasks":2,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/human-object-interaction-detection-on-hico","slug":"human-object-interaction-detection-on-hico","dataset":"HICO-DET","dataset_url":"/dataset/hico-det","rows_in_archive":55,"metrics":["mAP","Time Per Frame (ms)","Detection: Full (mAP@0.5)","Detection: Non-Rare (mAP@0.5)","Detection: Rare (mAP@0.5)"],"first_row_in_archive_order":{"model":"Ours (PViC+)","paper_title":"Dynamic Scene Understanding from Vision-Language Representations","paper_url":"/paper/dynamic-scene-understanding-from-vision","paper_date":"2025-01-20","arxiv_id":"2501.11653","code_links":[],"syntology":null}},{"leaderboard":"/sota/human-object-interaction-detection-on-v-coco","slug":"human-object-interaction-detection-on-v-coco","dataset":"V-COCO","dataset_url":"/dataset/v-coco","rows_in_archive":34,"metrics":["AP(S1)","AP(S2)","Time Per Frame(ms)","MAP"],"first_row_in_archive_order":{"model":"RLIPv2","paper_title":"RLIPv2: Fast Scaling of Relational Language-Image Pre-training","paper_url":"/paper/rlipv2-fast-scaling-of-relational-language","paper_date":"2023-08-18","arxiv_id":"2308.09351","code_links":[{"title":"jacobyuan7/rlipv2","url":"https://github.com/jacobyuan7/rlipv2"},{"title":"jacobyuan7/rlip","url":"https://github.com/jacobyuan7/rlip"},{"title":"jacobyuan7/ocn-hoi-benchmark","url":"https://github.com/jacobyuan7/ocn-hoi-benchmark"}],"syntology":{"n":30,"n_ran":22,"n_unverified":8,"n_pointer_only":0}}},{"leaderboard":"/sota/human-object-interaction-detection-on-hico-1","slug":"human-object-interaction-detection-on-hico-1","dataset":"HICO","dataset_url":"/dataset/hico","rows_in_archive":8,"metrics":["mAP"],"first_row_in_archive_order":{"model":"DEFR","paper_title":"The Overlooked Classifier in Human-Object Interaction Recognition","paper_url":"/paper/decoupling-object-detection-from-human-object-1","paper_date":"2021-12-13","arxiv_id":"2112.06392","code_links":[],"syntology":null}},{"leaderboard":"/sota/human-object-interaction-detection-on-vidhoi","slug":"human-object-interaction-detection-on-vidhoi","dataset":"VidHOI","dataset_url":"/dataset/vidhoi","rows_in_archive":3,"metrics":["Detection: Full (mAP@0.5)","Detection: Non-Rare (mAP@0.5)","Detection: Rare (mAP@0.5)","Oracle: Full (mAP@0.5)","Oracle: Non-Rare (mAP@0.5)","Oracle: Rare (mAP@0.5)"],"first_row_in_archive_order":{"model":"HOI4ABOT","paper_title":"HOI4ABOT: Human-Object Interaction Anticipation for Human Intention Reading Collaborative roBOTs","paper_url":"/paper/hoi4abot-human-object-interaction","paper_date":"2023-09-28","arxiv_id":"2309.16524","code_links":[],"syntology":null}},{"leaderboard":"/sota/human-object-interaction-detection-on","slug":"human-object-interaction-detection-on","dataset":"Ambiguious-HOI","dataset_url":null,"rows_in_archive":2,"metrics":["mAP"],"first_row_in_archive_order":{"model":"DJ-RN","paper_title":"Detailed 2D-3D Joint Representation for Human-Object Interaction","paper_url":"/paper/detailed-2d-3d-joint-representation-for-human","paper_date":"2020-04-17","arxiv_id":"2004.08154","code_links":[{"title":"DirtyHarryLYL/DJ-RN","url":"https://github.com/DirtyHarryLYL/DJ-RN"}],"syntology":{"n":12,"n_ran":0,"n_unverified":12,"n_pointer_only":0}}},{"leaderboard":"/sota/human-object-interaction-detection-on-meccano","slug":"human-object-interaction-detection-on-meccano","dataset":"MECCANO","dataset_url":"/dataset/meccano","rows_in_archive":1,"metrics":["mAP@0.5 role"],"first_row_in_archive_order":{"model":"SlowFast + FasterRCNN","paper_title":"The MECCANO Dataset: Understanding Human-Object Interactions from Egocentric Videos in an Industrial-like Domain","paper_url":"/paper/the-meccano-dataset-understanding-human","paper_date":"2020-10-12","arxiv_id":"2010.05654","code_links":[{"title":"fpv-iplab/MECCANO","url":"https://github.com/fpv-iplab/MECCANO"}],"syntology":null}}],"datasets":[{"url":"/dataset/hico-det","name":"HICO-DET","full_name":"HICO-DET","num_papers_in_archive":180},{"url":"/dataset/v-coco","name":"V-COCO","full_name":"Verbs in COCO","num_papers_in_archive":159},{"url":"/dataset/finegym","name":"FineGym","full_name":"FineGym","num_papers_in_archive":76},{"url":"/dataset/behave","name":"BEHAVE","full_name":"","num_papers_in_archive":53},{"url":"/dataset/rich","name":"RICH","full_name":"Real scenes, Interaction, Contact and Humans","num_papers_in_archive":53},{"url":"/dataset/hico","name":"HICO","full_name":"Humans Interacting with Common Objects","num_papers_in_archive":45},{"url":"/dataset/hoi4d","name":"HOI4D","full_name":"","num_papers_in_archive":23},{"url":"/dataset/meccano","name":"MECCANO","full_name":"","num_papers_in_archive":19},{"url":"/dataset/hake-large","name":"HAKE","full_name":"","num_papers_in_archive":15},{"url":"/dataset/watch-n-patch","name":"Watch-n-Patch","full_name":"Watch-n-Patch","num_papers_in_archive":13},{"url":"/dataset/couch","name":"COUCH","full_name":"","num_papers_in_archive":8},{"url":"/dataset/vidhoi","name":"VidHOI","full_name":"","num_papers_in_archive":7},{"url":"/dataset/chairs-dataset","name":"CHAIRS dataset","full_name":"","num_papers_in_archive":3},{"url":"/dataset/v-hico","name":"V-HICO","full_name":"","num_papers_in_archive":3},{"url":"/dataset/ambiguous-hoi","name":"Ambiguous-HOI","full_name":"","num_papers_in_archive":2},{"url":"/dataset/mphoi-72","name":"MPHOI-72","full_name":"Multi-person Human-object Interaction Dataset 72","num_papers_in_archive":2},{"url":"/dataset/darai","name":"DARai","full_name":"Daily Activity Recordings for AI and ML applications","num_papers_in_archive":1},{"url":"/dataset/dio","name":"DIO","full_name":"Discovering Interacted Objects","num_papers_in_archive":1},{"url":"/dataset/egoism-hoi","name":"EgoISM-HOI","full_name":"","num_papers_in_archive":1},{"url":"/dataset/h2o","name":"H2O","full_name":"","num_papers_in_archive":1},{"url":"/dataset/trek-100","name":"TREK-100","full_name":"","num_papers_in_archive":1},{"url":"/dataset/win-fail-action-understanding","name":"Win-Fail Action Understanding","full_name":"Win-Fail Action Understanding","num_papers_in_archive":1},{"url":"/dataset/h2o-interaction","name":"H²O Interaction","full_name":"Human-to-Human-or-Object Interaction","num_papers_in_archive":0}],"subtasks":[{"url":"/task/affordance-recognition","name":"Affordance Recognition"},{"url":"/task/hand-object-interaction-detection","name":"Hand-Object Interaction Detection"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":173,"tagged_in_all":449,"items":[{"url":"/paper/temporal-relational-reasoning-in-videos","title":"Temporal Relational Reasoning in Videos","date":"2017-11-22","arxiv_id":"1711.08496","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/visual-compositional-learning-for-human","title":"Visual Compositional Learning for Human-Object Interaction Detection","date":"2020-07-24","arxiv_id":"2007.12407","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/visual-semantic-graph-attention-network-for","title":"Visual-Semantic Graph Attention Networks for Human-Object Interaction Detection","date":"2020-01-07","arxiv_id":"2001.02302","repositories_listed":4,"syntology":null},{"url":"/paper/hake-human-activity-knowledge-engine","title":"HAKE: Human Activity Knowledge Engine","date":"2019-04-13","arxiv_id":"1904.06539","repositories_listed":4,"syntology":null},{"url":"/paper/ican-instance-centric-attention-network-for","title":"iCAN: Instance-Centric Attention Network for Human-Object Interaction Detection","date":"2018-08-30","arxiv_id":"1808.10437","repositories_listed":4,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/rlipv2-fast-scaling-of-relational-language","title":"RLIPv2: Fast Scaling of Relational Language-Image Pre-training","date":"2023-08-18","arxiv_id":"2308.09351","repositories_listed":3,"syntology":{"n":30,"n_ran":22,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/rlip-relational-language-image-pre-training","title":"RLIP: Relational Language-Image Pre-training for Human-Object Interaction Detection","date":"2022-09-05","arxiv_id":"2209.01814","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/hake-a-knowledge-engine-foundation-for-human","title":"HAKE: A Knowledge Engine Foundation for Human Activity Understanding","date":"2022-02-14","arxiv_id":"2202.06851","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/use-the-force-luke-learning-to-predict","title":"Use the Force, Luke! Learning to Predict Physical Forces by Simulating Effects","date":"2020-03-26","arxiv_id":"2003.12045","repositories_listed":3,"syntology":null},{"url":"/paper/transferable-interactiveness-prior-for-human","title":"Transferable Interactiveness Knowledge for Human-Object Interaction Detection","date":"2018-11-20","arxiv_id":"1811.08264","repositories_listed":3,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/no-frills-human-object-interaction-detection","title":"No-Frills Human-Object Interaction Detection: Factorization, Layout Encodings, and Training Techniques","date":"2018-11-14","arxiv_id":"1811.05967","repositories_listed":3,"syntology":null},{"url":"/paper/inject-semantic-concepts-into-image-tagging","title":"Open-Set Image Tagging with Multi-Grained Text Supervision","date":"2023-10-23","arxiv_id":"2310.15200","repositories_listed":2,"syntology":{"n":8,"n_ran":4,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/deco-dense-estimation-of-3d-human-scene-1","title":"DECO: Dense Estimation of 3D Human-Scene Contact In The Wild","date":"2023-09-26","arxiv_id":"2309.15273","repositories_listed":2,"syntology":{"n":12,"n_ran":7,"n_unverified":5,"n_pointer_only":12}},{"url":"/paper/focusing-on-what-to-decode-and-what-to-train","title":"Focusing on what to decode and what to train: SOV Decoding with Specific Target Guided DeNoising and Vision Language Advisor","date":"2023-07-05","arxiv_id":"2307.02291","repositories_listed":2,"syntology":null},{"url":"/paper/grounded-affordance-from-exocentric-view","title":"Grounded Affordance from Exocentric View","date":"2022-08-28","arxiv_id":"2208.13196","repositories_listed":2,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/discovering-human-object-interaction-concepts","title":"Discovering Human-Object Interaction Concepts via Self-Compositional Learning","date":"2022-03-27","arxiv_id":"2203.14272","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/learning-affordance-grounding-from-exocentric","title":"Learning Affordance Grounding from Exocentric Images","date":"2022-03-18","arxiv_id":"2203.09905","repositories_listed":2,"syntology":null},{"url":"/paper/d3d-hoi-dynamic-3d-human-object-interactions","title":"D3D-HOI: Dynamic 3D Human-Object Interactions from Videos","date":"2021-08-19","arxiv_id":"2108.08420","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/affordance-transfer-learning-for-human-object","title":"Affordance Transfer Learning for Human-Object Interaction Detection","date":"2021-04-07","arxiv_id":"2104.02867","repositories_listed":2,"syntology":null},{"url":"/paper/spatio-attentive-graphs-for-human-object","title":"Spatially Conditioned Graphs for Detecting Human-Object Interactions","date":"2020-12-11","arxiv_id":"2012.06060","repositories_listed":2,"syntology":null},{"url":"/paper/hoi-analysis-integrating-and-decomposing","title":"HOI Analysis: Integrating and Decomposing Human-Object Interaction","date":"2020-10-30","arxiv_id":"2010.16219","repositories_listed":2,"syntology":null},{"url":"/paper/consnet-learning-consistency-graph-for-zero","title":"ConsNet: Learning Consistency Graph for Zero-Shot Human-Object Interaction Detection","date":"2020-08-14","arxiv_id":"2008.06254","repositories_listed":2,"syntology":null},{"url":"/paper/polysemy-deciphering-network-for-robust-human","title":"Polysemy Deciphering Network for Robust Human-Object Interaction Detection","date":"2020-08-07","arxiv_id":"2008.02918","repositories_listed":2,"syntology":null},{"url":"/paper/perceiving-3d-human-object-spatial","title":"Perceiving 3D Human-Object Spatial Arrangements from a Single Image in the Wild","date":"2020-07-30","arxiv_id":"2007.15649","repositories_listed":2,"syntology":null},{"url":"/paper/pastanet-toward-human-activity-knowledge","title":"PaStaNet: Toward Human Activity Knowledge Engine","date":"2020-04-02","arxiv_id":"2004.00945","repositories_listed":2,"syntology":null},{"url":"/paper/vsgnet-spatial-attention-network-for","title":"VSGNet: Spatial Attention Network for Detecting Human Object Interactions Using Graph Convolutions","date":"2020-03-11","arxiv_id":"2003.05541","repositories_listed":2,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/detecting-and-recognizing-human-object","title":"Detecting and Recognizing Human-Object Interactions","date":"2017-04-24","arxiv_id":"1704.07333","repositories_listed":2,"syntology":null},{"url":"/paper/contextual-action-recognition-with-rcnn","title":"Contextual Action Recognition with R*CNN","date":"2015-05-05","arxiv_id":"1505.01197","repositories_listed":2,"syntology":null},{"url":"/paper/rohoi-robustness-benchmark-for-human-object","title":"RoHOI: Robustness Benchmark for Human-Object Interaction Detection","date":"2025-07-12","arxiv_id":"2507.09111","repositories_listed":1,"syntology":null},{"url":"/paper/bilateral-collaboration-with-large-vision","title":"Bilateral Collaboration with Large Vision-Language Models for Open Vocabulary Human-Object Interaction Detection","date":"2025-07-09","arxiv_id":"2507.06510","repositories_listed":1,"syntology":null}],"syntology_records":13,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}