{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/human-object-interaction-detection/papers/4","list_of":"/task/human-object-interaction-detection","task":"Human-Object Interaction Detection","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":5,"rows_per_page":100,"rows":[301,400],"of":449,"counts":{"archive_papers_tagged":449,"with_a_code_link":173,"where_syntology_ran_a_sample":65,"not_listed_spam_title":0,"listed":449,"listed_where_code_ran":65,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":58,"every_run_a_failure_of_syntologys_instrument":7,"listed_with_a_run_with_no_instrument_failure":58,"listed_every_run_a_failure_of_syntologys_instrument":7,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/human-object-interaction-detection","prev":"/task/human-object-interaction-detection/papers/3","next":"/task/human-object-interaction-detection/papers/5","papers":[{"url":null,"slug":"generating-human-centric-visual-cues-for","title":"Generating Human-Centric Visual Cues for Human-Object Interaction Detection via Large Vision-Language Models","date":"2023-11-26","arxiv_id":"2311.16475","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-unified-transformer-based-framework","title":"Towards a Unified Transformer-based Framework for Scene Graph Generation and Human-object Interaction Detection","date":"2023-11-03","arxiv_id":"2311.01755","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-generation-of-human-object-1","title":"Hierarchical Generation of Human-Object Interactions with Diffusion Probabilistic Models","date":"2023-10-03","arxiv_id":"2310.02242","repositories_listed":0,"syntology":null},{"url":"/paper/hoi4abot-human-object-interaction","slug":"hoi4abot-human-object-interaction","title":"HOI4ABOT: Human-Object Interaction Anticipation for Human Intention Reading Collaborative roBOTs","date":"2023-09-28","arxiv_id":"2309.16524","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-motion-guided-human-motion-synthesis","title":"Object Motion Guided Human Motion Synthesis","date":"2023-09-28","arxiv_id":"2309.16237","repositories_listed":0,"syntology":null},{"url":null,"slug":"detection-and-localization-of-firearm","title":"Detection and Localization of Firearm Carriers in Complex Scenes for Improved Safety Measures","date":"2023-09-17","arxiv_id":"2309.09236","repositories_listed":0,"syntology":null},{"url":null,"slug":"physically-plausible-full-body-hand-object","title":"Physically Plausible Full-Body Hand-Object Interaction Synthesis","date":"2023-09-14","arxiv_id":"2309.07907","repositories_listed":0,"syntology":null},{"url":null,"slug":"chorus-learning-canonicalized-3d-human-object","title":"CHORUS: Learning Canonicalized 3D Human-Object Spatial Relations from Unbounded Synthesized Images","date":"2023-08-23","arxiv_id":"2308.12288","repositories_listed":0,"syntology":null},{"url":null,"slug":"hodn-disentangling-human-object-feature-for","title":"HODN: Disentangling Human-Object Feature for HOI Detection","date":"2023-08-20","arxiv_id":"2308.10158","repositories_listed":0,"syntology":null},{"url":null,"slug":"agglomerative-transformer-for-human-object","title":"Agglomerative Transformer for Human-Object Interaction Detection","date":"2023-08-16","arxiv_id":"2308.08370","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-next-active-objects-for-context","title":"Leveraging Next-Active Objects for Context-Aware Anticipation in Egocentric Videos","date":"2023-08-16","arxiv_id":"2308.08303","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-learning-in-transformer-based","title":"Compositional Learning in Transformer-Based Human-Object Interaction Detection","date":"2023-08-11","arxiv_id":"2308.05961","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-human-object-interaction-detection-1","title":"Improving Human-Object Interaction Detection via Virtual Image Learning","date":"2023-08-04","arxiv_id":"2308.02606","repositories_listed":0,"syntology":null},{"url":null,"slug":"dpmix-mixture-of-depth-and-point-cloud-video","title":"DPMix: Mixture of Depth and Point Cloud Video Experts for 4D Action Segmentation","date":"2023-07-31","arxiv_id":"2307.16803","repositories_listed":0,"syntology":null},{"url":null,"slug":"re-mine-learn-and-reason-exploring-the-cross","title":"Re-mine, Learn and Reason: Exploring the Cross-modal Semantic Correlations for Language-guided HOI detection","date":"2023-07-25","arxiv_id":"2307.13529","repositories_listed":0,"syntology":null},{"url":null,"slug":"persistent-transient-duality-a-multi","title":"Persistent-Transient Duality: A Multi-mechanism Approach for Modeling Human-Object Interaction","date":"2023-07-24","arxiv_id":"2307.12729","repositories_listed":0,"syntology":null},{"url":null,"slug":"mining-conditional-part-semantics-with","title":"Mining Conditional Part Semantics with Occluded Extrapolation for Human-Object Interaction Detection","date":"2023-07-19","arxiv_id":"2307.10499","repositories_listed":0,"syntology":null},{"url":null,"slug":"hokem-human-and-object-keypoint-based","title":"HOKEM: Human and Object Keypoint-based Extension Module for Human-Object Interaction Detection","date":"2023-06-25","arxiv_id":"2306.14260","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-abstract-specification-of-voxml-as-an","title":"An Abstract Specification of VoxML as an Annotation Language","date":"2023-05-22","arxiv_id":"2305.13076","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-diverse-human-motions-in-3d","title":"Synthesizing Diverse Human Motions in 3D Indoor Scenes","date":"2023-05-21","arxiv_id":"2305.12411","repositories_listed":0,"syntology":null},{"url":null,"slug":"group-activity-recognition-via-dynamic","title":"Group Activity Recognition via Dynamic Composition and Interaction","date":"2023-05-09","arxiv_id":"2305.05583","repositories_listed":0,"syntology":null},{"url":null,"slug":"modelling-spatio-temporal-interactions-for","title":"Modelling Spatio-Temporal Interactions for Compositional Action Recognition","date":"2023-05-04","arxiv_id":"2305.02673","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-3d-human-object-neural","title":"Compositional 3D Human-Object Neural Animation","date":"2023-04-27","arxiv_id":"2304.14070","repositories_listed":0,"syntology":null},{"url":null,"slug":"hosnerf-dynamic-human-object-scene-neural","title":"HOSNeRF: Dynamic Human-Object-Scene Neural Radiance Fields from a Single Video","date":"2023-04-24","arxiv_id":"2304.12281","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-based-contrastive-learning-on-decision","title":"Video-based Contrastive Learning on Decision Trees: from Action Recognition to Autism Diagnosis","date":"2023-04-20","arxiv_id":"2304.10073","repositories_listed":0,"syntology":null},{"url":null,"slug":"instant-nvr-instant-neural-volumetric","title":"Instant-NVR: Instant Neural Volumetric Rendering for Human-object Interactions from Monocular RGBD Stream","date":"2023-04-06","arxiv_id":"2304.03184","repositories_listed":0,"syntology":null},{"url":null,"slug":"visibility-aware-human-object-interaction","title":"Visibility Aware Human-Object Interaction Tracking from Single RGB Camera","date":"2023-03-29","arxiv_id":"2303.16479","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-human-object-interactions","title":"Task-Oriented Human-Object Interactions Generation with Implicit Neural Representations","date":"2023-03-23","arxiv_id":"2303.13129","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-hoi-detection-from","title":"Weakly-Supervised HOI Detection from Interaction Labels Only and Language/Vision-Language Priors","date":"2023-03-09","arxiv_id":"2303.05546","repositories_listed":0,"syntology":null},{"url":null,"slug":"skghoi-spatial-semantic-knowledge-graph-for","title":"TMHOI: Translational Model for Human-Object Interaction Detection","date":"2023-03-07","arxiv_id":"2303.04253","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-hoi-detection-via-prior","title":"Weakly-supervised HOI Detection via Prior-guided Bi-level Representation Learning","date":"2023-03-02","arxiv_id":"2303.01313","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallel-reasoning-network-for-human-object","title":"Parallel Reasoning Network for Human-Object Interaction Detection","date":"2023-01-09","arxiv_id":"2301.03510","repositories_listed":0,"syntology":null},{"url":null,"slug":"chorus-learning-canonicalized-3d-human-object-1","title":"CHORUS : Learning Canonicalized 3D Human-Object Spatial Relations from Unbounded Synthesized Images","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dropkey-for-vision-transformer","title":"DropKey for Vision Transformer","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"open-category-human-object-interaction-pre","title":"Open-Category Human-Object Interaction Pre-Training via Language Modeling Framework","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"open-set-video-hoi-detection-from-action","title":"Open Set Video HOI detection from Action-Centric Chain-of-Look Prompting","date":"2023-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"neuraldome-a-neural-modeling-pipeline-on","title":"NeuralDome: A Neural Modeling Pipeline on Multi-View Human-Object Interactions","date":"2022-12-15","arxiv_id":"2212.07626","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-state-change-capture-of","title":"Exploring State Change Capture of Heterogeneous Backbones @ Ego4D Hands and Objects Challenge 2022","date":"2022-11-16","arxiv_id":"2211.08728","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-object-tracking-in-first-person-vision","title":"Visual Object Tracking in First Person Vision","date":"2022-09-27","arxiv_id":"2209.13502","repositories_listed":0,"syntology":null},{"url":null,"slug":"meccano-a-multimodal-egocentric-dataset-for","title":"MECCANO: A Multimodal Egocentric Dataset for Humans Behavior Understanding in the Industrial-like Domain","date":"2022-09-19","arxiv_id":"2209.08691","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphing-the-future-activity-and-next-active","title":"Graphing the Future: Activity and Next Active Object Prediction using Graph-based Activity Representations","date":"2022-09-12","arxiv_id":"2209.05194","repositories_listed":0,"syntology":null},{"url":null,"slug":"reconstructing-action-conditioned-human","title":"Reconstructing Action-Conditioned Human-Object Interactions Using Commonsense Knowledge Priors","date":"2022-09-06","arxiv_id":"2209.02485","repositories_listed":0,"syntology":null},{"url":null,"slug":"dropkey","title":"DropKey","date":"2022-08-04","arxiv_id":"2208.02646","repositories_listed":0,"syntology":null},{"url":null,"slug":"savchoi-detecting-suspicious-activities-using","title":"SAVCHOI: Detecting Suspicious Activities using Dense Video Captioning with Human Object Interactions","date":"2022-07-24","arxiv_id":"2207.11838","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-guided-bidirectional-attention","title":"Knowledge Guided Bidirectional Attention Network for Human-Object Interaction Detection","date":"2022-07-16","arxiv_id":"2207.07979","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-structured-representations-of-visual","title":"Learning Structured Representations of Visual Scenes","date":"2022-07-09","arxiv_id":"2207.04200","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-predict-how-humans-manipulate-large","title":"Learn to Predict How Humans Manipulate Large-sized Objects from Interactive Motions","date":"2022-06-25","arxiv_id":"2206.12612","repositories_listed":0,"syntology":null},{"url":null,"slug":"precise-affordance-annotation-for-egocentric","title":"Precise Affordance Annotation for Egocentric Action Video Datasets","date":"2022-06-11","arxiv_id":"2206.05424","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-parsing-and-dynamic-temporal-pooling","title":"Spatial Parsing and Dynamic Temporal Pooling networks for Human-Object Interaction detection","date":"2022-06-07","arxiv_id":"2206.03061","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-based-human-object-interaction","title":"Video-based Human-Object Interaction Detection from Tubelet Tokens","date":"2022-06-04","arxiv_id":"2206.01908","repositories_listed":0,"syntology":null},{"url":null,"slug":"visually-plausible-human-object-interaction","title":"Interaction Replica: Tracking Human-Object Interaction and Scene Changes From Human Motion","date":"2022-05-05","arxiv_id":"2205.02830","repositories_listed":0,"syntology":null},{"url":"/paper/couch-towards-controllable-human-chair","slug":"couch-towards-controllable-human-chair","title":"COUCH: Towards Controllable Human-Chair Interactions","date":"2022-05-01","arxiv_id":"2205.00541","repositories_listed":0,"syntology":null},{"url":null,"slug":"persistent-transient-duality-in-human","title":"Persistent-Transient Duality in Human Behavior Modeling","date":"2022-04-21","arxiv_id":"2204.09875","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-object-interaction-detection-via","title":"Human-Object Interaction Detection via Disentangled Transformer","date":"2022-04-20","arxiv_id":"2204.09290","repositories_listed":0,"syntology":null},{"url":null,"slug":"thorn-temporal-human-object-relation-network","title":"THORN: Temporal Human-Object Relation Network for Action Recognition","date":"2022-04-20","arxiv_id":"2204.09468","repositories_listed":0,"syntology":null},{"url":null,"slug":"category-aware-transformer-network-for-better","title":"Category-Aware Transformer Network for Better Human-Object Interaction Detection","date":"2022-04-11","arxiv_id":"2204.04911","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-to-look-at-and-where-semantic-and","title":"What to look at and where: Semantic and Spatial Refined Transformer for detecting human-object interactions","date":"2022-04-02","arxiv_id":"2204.00746","repositories_listed":0,"syntology":null},{"url":null,"slug":"mstr-multi-scale-transformer-for-end-to-end","title":"MSTR: Multi-Scale Transformer for End-to-End Human-Object Interaction Detection","date":"2022-03-28","arxiv_id":"2203.14709","repositories_listed":0,"syntology":null},{"url":null,"slug":"iwin-human-object-interaction-detection-via","title":"Iwin: Human-Object Interaction Detection via Transformer with Irregular Windows","date":"2022-03-20","arxiv_id":"2203.10537","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-overlooked-classifier-in-human-object","title":"The Overlooked Classifier in Human-Object Interaction Recognition","date":"2022-03-10","arxiv_id":"2203.05676","repositories_listed":0,"syntology":null},{"url":null,"slug":"neuralfusion-neural-volumetric-rendering","title":"NeuralHOFusion: Neural Volumetric Rendering under Human-object Interactions","date":"2022-02-25","arxiv_id":"2202.12825","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-actor-centric-human-object","title":"Effective Actor-centric Human-object Interaction Detection","date":"2022-02-24","arxiv_id":"2202.11998","repositories_listed":0,"syntology":null},{"url":"/paper/webly-supervised-concept-expansion-for","slug":"webly-supervised-concept-expansion-for","title":"Webly Supervised Concept Expansion for General Purpose Vision Models","date":"2022-02-04","arxiv_id":"2202.02317","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-stage-deep-transfer-learning-for-emiot","title":"Multi-Stage Deep Transfer Learning for EmIoT-enabled Human-Computer Interaction","date":"2022-02-03","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dexvip-learning-dexterous-grasping-with-human","title":"DexVIP: Learning Dexterous Grasping with Human Hand Pose Priors from Video","date":"2022-02-01","arxiv_id":"2202.00164","repositories_listed":0,"syntology":null},{"url":null,"slug":"complex-video-action-reasoning-via-learnable","title":"Complex Video Action Reasoning via Learnable Markov Logic Network","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distillation-using-oracle-queries-for","title":"Distillation Using Oracle Queries for Transformer-Based Human-Object Interaction Detection","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"expansion-squeeze-excitation-fusion-network","title":"Expansion-Squeeze-Excitation Fusion Network for Elderly Activity Recognition","date":"2021-12-21","arxiv_id":"2112.10992","repositories_listed":0,"syntology":null},{"url":null,"slug":"distillation-of-human-object-interaction","title":"Distillation of Human-Object Interaction Contexts for Action Recognition","date":"2021-12-17","arxiv_id":"2112.09448","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-human-object-interaction-detection","title":"Improving Human-Object Interaction Detection via Phrase Learning and Label Composition","date":"2021-12-14","arxiv_id":"2112.07383","repositories_listed":0,"syntology":null},{"url":"/paper/decoupling-object-detection-from-human-object-1","slug":"decoupling-object-detection-from-human-object-1","title":"The Overlooked Classifier in Human-Object Interaction Recognition","date":"2021-12-13","arxiv_id":"2112.06392","repositories_listed":0,"syntology":null},{"url":null,"slug":"demograsp-few-shot-learning-for-robotic","title":"DemoGrasp: Few-Shot Learning for Robotic Grasping with Human Demonstration","date":"2021-12-06","arxiv_id":"2112.02849","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-object-interaction-detection-via-weak","title":"Human-Object Interaction Detection via Weak Supervision","date":"2021-12-01","arxiv_id":"2112.00492","repositories_listed":0,"syntology":null},{"url":null,"slug":"estimating-3d-motion-and-forces-of-human","title":"Estimating 3D Motion and Forces of Human-Object Interactions from Internet Videos","date":"2021-11-02","arxiv_id":"2111.01591","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-spatio-temporal-identity-verification","title":"whu-nercms at trecvid2021:instance search task","date":"2021-10-30","arxiv_id":"2111.00228","repositories_listed":0,"syntology":null},{"url":"/paper/is-first-person-vision-challenging-for-object-1","slug":"is-first-person-vision-challenging-for-object-1","title":"Is First Person Vision Challenging for Object Tracking?","date":"2021-08-31","arxiv_id":"2108.13665","repositories_listed":0,"syntology":null},{"url":null,"slug":"gravity-aware-monocular-3d-human-object","title":"Gravity-Aware Monocular 3D Human-Object Reconstruction","date":"2021-08-19","arxiv_id":"2108.08844","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatio-temporal-interaction-graph-parsing","title":"Spatio-Temporal Interaction Graph Parsing Networks for Human-Object Interaction Recognition","date":"2021-08-19","arxiv_id":"2108.08633","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-free-viewpoint-performance-rendering","title":"Neural Free-Viewpoint Performance Rendering under Complex Human-object Interactions","date":"2021-08-01","arxiv_id":"2108.00362","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-object-detection-necessary-for-human-1","title":"Is Object Detection Necessary for Human-Object Interaction Recognition?","date":"2021-07-27","arxiv_id":"2107.13083","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-video-prediction-using","title":"Hierarchical Video Prediction Using Relational Layouts for Human-Object Interactions","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-guided-segmentation-framework-for","title":"Contextual Guided Segmentation Framework for Semi-supervised Video Instance Segmentation","date":"2021-06-07","arxiv_id":"2106.03330","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-object-interaction-detection-using-two","title":"Human Object Interaction Detection using Two-Direction Spatial Enhancement and Exclusive Object Prior","date":"2021-05-07","arxiv_id":"2105.03089","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-composite-set-detection-using-part-and","title":"Visual Relationship Detection Using Part-and-Sum Transformers with Composite Queries","date":"2021-05-05","arxiv_id":"2105.02170","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustfusion-robust-volumetric-performance","title":"RobustFusion: Robust Volumetric Performance Reconstruction under Human-object Interactions from Monocular RGBD Stream","date":"2021-04-30","arxiv_id":"2104.14837","repositories_listed":0,"syntology":null},{"url":null,"slug":"rr-net-injecting-interactive-semantics-in","title":"RR-Net: Injecting Interactive Semantics in Human-Object Interaction Detection","date":"2021-04-30","arxiv_id":"2104.15015","repositories_listed":0,"syntology":null},{"url":null,"slug":"tripod-human-trajectory-and-pose-dynamics","title":"TRiPOD: Human Trajectory and Pose Dynamics Forecasting in the Wild","date":"2021-04-08","arxiv_id":"2104.04029","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-asynchronous-and-sparse-human-object","title":"Learning Asynchronous and Sparse Human-Object Interaction in Videos","date":"2021-03-03","arxiv_id":"2103.02758","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-human-interactions-with-large","title":"Discovering Human Interactions With Large-Vocabulary Objects via Query and Multi-Scale Detection","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-overcoming-false-positives-in-visual","title":"Towards Overcoming False Positives in Visual Relationship Detection","date":"2020-12-23","arxiv_id":"2012.12510","repositories_listed":0,"syntology":null},{"url":"/paper/is-first-person-vision-challenging-for-object","slug":"is-first-person-vision-challenging-for-object","title":"Is First Person Vision Challenging for Object Tracking?","date":"2020-11-24","arxiv_id":"2011.12263","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-human-object-interaction-with-mixed","title":"Detecting Human-Object Interaction with Mixed Supervision","date":"2020-11-10","arxiv_id":"2011.04971","repositories_listed":0,"syntology":null},{"url":null,"slug":"kinematics-guided-reinforcement-learning-for","title":"Kinematics-Guided Reinforcement Learning for Object-Aware 3D Ego-Pose Estimation","date":"2020-11-10","arxiv_id":"2011.04837","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-heterogeneous-graph-network-for-1","title":"Contextual Heterogeneous Graph Network for Human-Object Interaction Detection","date":"2020-10-20","arxiv_id":"2010.10001","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-selective-context-for-interaction","title":"Self-Selective Context for Interaction Recognition","date":"2020-10-17","arxiv_id":"2010.08750","repositories_listed":0,"syntology":null},{"url":"/paper/decaug-augmenting-hoi-detection-via","slug":"decaug-augmenting-hoi-detection-via","title":"DecAug: Augmenting HOI Detection via Decomposition","date":"2020-10-02","arxiv_id":"2010.01007","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-human-object-interaction","title":"Zero-Shot Human-Object Interaction Recognition via Affordance Graphs","date":"2020-09-02","arxiv_id":"2009.01039","repositories_listed":0,"syntology":null},{"url":null,"slug":"amplifying-key-cues-for-human-object","title":"Amplifying Key Cues for Human-Object-Interaction Detection","date":"2020-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-graph-based-interactive-reasoning-for-human","title":"A Graph-based Interactive Reasoning for Human-Object Interaction Detection","date":"2020-07-14","arxiv_id":"2007.06925","repositories_listed":0,"syntology":null},{"url":null,"slug":"cobe-contextualized-object-embeddings-from","title":"COBE: Contextualized Object Embeddings from Narrated Instructional Video","date":"2020-07-14","arxiv_id":"2007.07306","repositories_listed":0,"syntology":null}],"record_sha256":"4009c222db4dd332321feb52a5c900217ad95637397334e8196536dc9c971583","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}