{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/object/papers/42","list_of":"/task/object","task":"Object","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":42,"pages_in_order":107,"rows_per_page":100,"rows":[4101,4200],"of":10696,"counts":{"archive_papers_tagged":10696,"with_a_code_link":3979,"where_syntology_ran_a_sample":1043,"not_listed_spam_title":0,"listed":10696,"listed_where_code_ran":1043,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":919,"every_run_a_failure_of_syntologys_instrument":124,"listed_with_a_run_with_no_instrument_failure":919,"listed_every_run_a_failure_of_syntologys_instrument":124,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/object","prev":"/task/object/papers/41","next":"/task/object/papers/43","papers":[{"url":null,"slug":"countdiffusion-text-to-image-synthesis-with","title":"CountDiffusion: Text-to-Image Synthesis with Training-Free Counting-Guidance Diffusion","date":"2025-05-07","arxiv_id":"2505.04347","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resolution-next-best-view-for-robot","title":"Low Resolution Next Best View for Robot Packing","date":"2025-05-07","arxiv_id":"2505.04228","repositories_listed":0,"syntology":null},{"url":"/paper/one2any-one-reference-6d-pose-estimation-for","slug":"one2any-one-reference-6d-pose-estimation-for","title":"One2Any: One-Reference 6D Pose Estimation for Any Object","date":"2025-05-07","arxiv_id":"2505.04109","repositories_listed":0,"syntology":{"n":19,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":17,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/one2any-one-reference-6d-pose-estimation-for#ran","syntology_url":"https://syntology.ai/paper/2505.04109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.04109"}},"official":null}},{"url":null,"slug":"web2grasp-learning-functional-grasps-from-web","title":"Web2Grasp: Learning Functional Grasps from Web Images of Hand-Object Interactions","date":"2025-05-07","arxiv_id":"2505.05517","repositories_listed":0,"syntology":null},{"url":null,"slug":"corner-cases-how-size-and-position-of-objects","title":"Corner Cases: How Size and Position of Objects Challenge ImageNet-Trained Models","date":"2025-05-06","arxiv_id":"2505.03569","repositories_listed":0,"syntology":null},{"url":null,"slug":"eopose-exemplar-based-object-reposing-using","title":"EOPose : Exemplar-based object reposing using Generalized Pose Correspondences","date":"2025-05-06","arxiv_id":"2505.03394","repositories_listed":0,"syntology":null},{"url":null,"slug":"null-counterfactual-factor-interactions-for","title":"Null Counterfactual Factor Interactions for Goal-Conditioned Reinforcement Learning","date":"2025-05-06","arxiv_id":"2505.03172","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-compact-clustering-attention","title":"Hierarchical Compact Clustering Attention (COCA) for Unsupervised Object-Centric Learning","date":"2025-05-04","arxiv_id":"2505.02071","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-interactive-3d-segmentation","title":"Probabilistic Interactive 3D Segmentation with Hierarchical Neural Processes","date":"2025-05-03","arxiv_id":"2505.01726","repositories_listed":0,"syntology":null},{"url":null,"slug":"resanything-attribute-prompting-for-arbitrary","title":"RESAnything: Attribute Prompting for Arbitrary Referring Segmentation","date":"2025-05-03","arxiv_id":"2505.02867","repositories_listed":0,"syntology":null},{"url":null,"slug":"freeinsert-disentangled-text-guided-object","title":"FreeInsert: Disentangled Text-Guided Object Insertion in 3D Gaussian Scene without Spatial Priors","date":"2025-05-02","arxiv_id":"2505.01322","repositories_listed":0,"syntology":null},{"url":null,"slug":"heal3d-heuristical-enhanced-active-learning","title":"HeAL3D: Heuristical-enhanced Active Learning for 3D Object Detection","date":"2025-05-01","arxiv_id":"2505.00507","repositories_listed":0,"syntology":null},{"url":null,"slug":"inconsistency-based-active-learning-for-lidar","title":"Inconsistency-based Active Learning for LiDAR Object Detection","date":"2025-05-01","arxiv_id":"2505.00511","repositories_listed":0,"syntology":null},{"url":null,"slug":"black-box-visual-prompt-engineering-for","title":"Black-Box Visual Prompt Engineering for Mitigating Object Hallucination in Large Vision Language Models","date":"2025-04-30","arxiv_id":"2504.21559","repositories_listed":0,"syntology":null},{"url":null,"slug":"dope-dual-object-perception-enhancement","title":"DOPE: Dual Object Perception-Enhancement Network for Vision-and-Language Navigation","date":"2025-04-30","arxiv_id":"2505.00743","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-self-supervised-fine-grained-video","title":"Enhancing Self-Supervised Fine-Grained Video Object Tracking with Dynamic Memory Prediction","date":"2025-04-30","arxiv_id":"2504.21692","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-borrow-features-for-improved","title":"Learning to Borrow Features for Improved Detection of Small Objects in Single-Shot Detectors","date":"2025-04-30","arxiv_id":"2505.00044","repositories_listed":0,"syntology":null},{"url":null,"slug":"mosam-motion-guided-segment-anything-model","title":"MoSAM: Motion-Guided Segment Anything Model with Spatial-Temporal Memory Selection","date":"2025-04-30","arxiv_id":"2505.00739","repositories_listed":0,"syntology":null},{"url":null,"slug":"stereo-x-ray-tomography-on-deformed-object","title":"Stereo X-ray tomography on deformed object tracking","date":"2025-04-30","arxiv_id":"2505.00122","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-mean-of-multi-object-trajectories","title":"The Mean of Multi-Object Trajectories","date":"2025-04-29","arxiv_id":"2504.20391","repositories_listed":0,"syntology":null},{"url":null,"slug":"category-level-and-open-set-object-pose","title":"Category-Level and Open-Set Object Pose Estimation for Robotics","date":"2025-04-28","arxiv_id":"2504.19572","repositories_listed":0,"syntology":null},{"url":null,"slug":"lm-mcvt-a-lightweight-multi-modal-multi-view","title":"LM-MCVT: A Lightweight Multi-modal Multi-view Convolutional-Vision Transformer Approach for 3D Object Recognition","date":"2025-04-27","arxiv_id":"2504.19256","repositories_listed":0,"syntology":null},{"url":null,"slug":"dexonomy-synthesizing-all-dexterous-grasp","title":"Dexonomy: Synthesizing All Dexterous Grasp Types in a Grasp Taxonomy","date":"2025-04-26","arxiv_id":"2504.18829","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-3d-object-detection-with-vision","title":"A Review of 3D Object Detection with Vision-Language Models","date":"2025-04-25","arxiv_id":"2504.18738","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-sensor-fusion-of-active-and-passive","title":"Multi-Sensor Fusion of Active and Passive Measurements for Extended Object Tracking","date":"2025-04-25","arxiv_id":"2504.18301","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-learning-and-robust-3d-reconstruction","title":"Object Learning and Robust 3D Reconstruction","date":"2025-04-22","arxiv_id":"2504.17812","repositories_listed":0,"syntology":null},{"url":null,"slug":"pcf-grasp-converting-point-completion-to","title":"PCF-Grasp: Converting Point Completion to Geometry Feature to Enhance 6-DoF Grasp","date":"2025-04-22","arxiv_id":"2504.16320","repositories_listed":0,"syntology":null},{"url":null,"slug":"deeppd-joint-phase-and-object-estimation-from","title":"DeepPD: Joint Phase and Object Estimation from Phase Diversity with Neural Calibration of a Deformable Mirror","date":"2025-04-19","arxiv_id":"2504.14157","repositories_listed":0,"syntology":null},{"url":null,"slug":"hmpe-heatmap-embedding-for-efficient","title":"HMPE:HeatMap Embedding for Efficient Transformer-Based Small Object Detection","date":"2025-04-18","arxiv_id":"2504.13469","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-intention-grounding-for-egocentric","title":"Visual Intention Grounding for Egocentric Assistants","date":"2025-04-18","arxiv_id":"2504.13621","repositories_listed":0,"syntology":null},{"url":null,"slug":"crossing-the-human-robot-embodiment-gap-with","title":"Crossing the Human-Robot Embodiment Gap with Sim-to-Real RL using One Human Demonstration","date":"2025-04-17","arxiv_id":"2504.12609","repositories_listed":0,"syntology":null},{"url":null,"slug":"hiscene-creating-hierarchical-3d-scenes-with","title":"HiScene: Creating Hierarchical 3D Scenes with Isometric View Generation","date":"2025-04-17","arxiv_id":"2504.13072","repositories_listed":0,"syntology":null},{"url":null,"slug":"rf-detr-object-detection-vs-yolov12-a-study","title":"RF-DETR Object Detection vs YOLOv12 : A Study of Transformer-based and CNN-based Architectures for Single-Class and Multi-Class Greenfruit Detection in Complex Orchard Environments Under Label Ambiguity","date":"2025-04-17","arxiv_id":"2504.13099","repositories_listed":0,"syntology":null},{"url":null,"slug":"sar-object-detection-with-self-supervised","title":"SAR Object Detection with Self-Supervised Pretraining and Curriculum-Aware Sampling","date":"2025-04-17","arxiv_id":"2504.13310","repositories_listed":0,"syntology":null},{"url":null,"slug":"vita-zero-zero-shot-visuotactile-object-6d","title":"ViTa-Zero: Zero-shot Visuotactile Object 6D Pose Estimation","date":"2025-04-17","arxiv_id":"2504.13179","repositories_listed":0,"syntology":null},{"url":null,"slug":"vllfl-a-vision-language-model-based","title":"VLLFL: A Vision-Language Model Based Lightweight Federated Learning Framework for Smart Agriculture","date":"2025-04-17","arxiv_id":"2504.13365","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-yolov12-attention-based","title":"A Review of YOLOv12: Attention-Based Enhancements vs. Previous Versions","date":"2025-04-16","arxiv_id":"2504.11995","repositories_listed":0,"syntology":null},{"url":null,"slug":"dm-osvp-one-shot-view-planning-using-3d","title":"DM-OSVP++: One-Shot View Planning Using 3D Diffusion Models for Active RGB-Based Object Reconstruction","date":"2025-04-16","arxiv_id":"2504.11674","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-visual-relation-detection-with","title":"Generalized Visual Relation Detection with Diffusion Models","date":"2025-04-16","arxiv_id":"2504.12100","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-placement-for-anything","title":"Object Placement for Anything","date":"2025-04-16","arxiv_id":"2504.12029","repositories_listed":0,"syntology":null},{"url":null,"slug":"radler-radar-object-detection-leveraging","title":"RADLER: Radar Object Detection Leveraging Semantic 3D City Models and Self-Supervised Radar-Image Learning","date":"2025-04-16","arxiv_id":"2504.12167","repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-advance-in-3d-object-and-scene","title":"Recent Advance in 3D Object and Scene Generation: A Survey","date":"2025-04-16","arxiv_id":"2504.11734","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-object-reconstruction-with-mmwave-radars","title":"3D Object Reconstruction with mmWave Radars","date":"2025-04-15","arxiv_id":"2504.12348","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-construct-redefining-construction-safety","title":"Safe-Construct: Redefining Construction Safety Violation Recognition as 3D Multi-View Engagement Task","date":"2025-04-15","arxiv_id":"2504.10880","repositories_listed":0,"syntology":null},{"url":null,"slug":"weather-aware-object-detection-transformer","title":"Weather-Aware Object Detection Transformer for Domain Adaptation","date":"2025-04-15","arxiv_id":"2504.10877","repositories_listed":0,"syntology":null},{"url":null,"slug":"counts-benchmarking-object-detectors-and","title":"COUNTS: Benchmarking Object Detectors and Multimodal Large Language Models under Distribution Shifts","date":"2025-04-14","arxiv_id":"2504.10158","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffmod-progressive-diffusion-point-denoising","title":"DiffMOD: Progressive Diffusion Point Denoising for Moving Object Detection in Remote Sensing","date":"2025-04-14","arxiv_id":"2504.10278","repositories_listed":0,"syntology":null},{"url":null,"slug":"humoto-a-4d-dataset-of-mocap-human-object","title":"HUMOTO: A 4D Dataset of Mocap Human Object Interactions","date":"2025-04-14","arxiv_id":"2504.10414","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-object-grounding-via-hierarchical","title":"Multi-Object Grounding via Hierarchical Contrastive Siamese Transformers","date":"2025-04-14","arxiv_id":"2504.10048","repositories_listed":0,"syntology":null},{"url":null,"slug":"riccardo-radar-hit-prediction-and-convolution","title":"RICCARDO: Radar Hit Prediction and Convolution for Camera-Radar 3D Object Detection","date":"2025-04-12","arxiv_id":"2504.09086","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-twin-catalog-a-large-scale","title":"Digital Twin Catalog: A Large-Scale Photorealistic 3D Object Digital Twin Dataset","date":"2025-04-11","arxiv_id":"2504.08541","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-free-guidance-in-text-to-video","title":"Training-free Guidance in Text-to-Video Generation via Multimodal Planning and Structured Noise Initialization","date":"2025-04-11","arxiv_id":"2504.08641","repositories_listed":0,"syntology":null},{"url":null,"slug":"boxdreamer-dreaming-box-corners-for","title":"BoxDreamer: Dreaming Box Corners for Generalizable Object Pose Estimation","date":"2025-04-10","arxiv_id":"2504.07955","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-can-objects-help-video-language","title":"How Can Objects Help Video-Language Understanding?","date":"2025-04-10","arxiv_id":"2504.07454","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-object-focused-attention","title":"Learning Object Focused Attention","date":"2025-04-10","arxiv_id":"2504.08166","repositories_listed":0,"syntology":null},{"url":null,"slug":"marmot-multi-agent-reasoning-for-multi-object","title":"Marmot: Multi-Agent Reasoning for Multi-Object Self-Correcting in Improving Image-Text Alignment","date":"2025-04-10","arxiv_id":"2504.20054","repositories_listed":0,"syntology":null},{"url":"/paper/poem-precise-object-level-editing-via-mllm","slug":"poem-precise-object-level-editing-via-mllm","title":"POEM: Precise Object-level Editing via MLLM control","date":"2025-04-10","arxiv_id":"2504.08111","repositories_listed":0,"syntology":null},{"url":null,"slug":"samjam-zero-shot-video-scene-graph-generation","title":"SAMJAM: Zero-Shot Video Scene Graph Generation for Egocentric Kitchen Videos","date":"2025-04-10","arxiv_id":"2504.07867","repositories_listed":0,"syntology":null},{"url":null,"slug":"ws-detr-robust-water-surface-object-detection","title":"WS-DETR: Robust Water Surface Object Detection through Vision-Radar Fusion with Detection Transformer","date":"2025-04-10","arxiv_id":"2504.07441","repositories_listed":0,"syntology":null},{"url":null,"slug":"better-decisions-through-the-right-causal","title":"Better Decisions through the Right Causal World Model","date":"2025-04-09","arxiv_id":"2504.07257","repositories_listed":0,"syntology":null},{"url":null,"slug":"compass-control-multi-object-orientation","title":"Compass Control: Multi Object Orientation Control for Text-to-Image Generation","date":"2025-04-09","arxiv_id":"2504.06752","repositories_listed":0,"syntology":null},{"url":null,"slug":"dltpose-6dof-pose-estimation-from-accurate","title":"DLTPose: 6DoF Pose Estimation From Accurate Dense Surface Point Estimates","date":"2025-04-09","arxiv_id":"2504.07335","repositories_listed":0,"syntology":null},{"url":null,"slug":"glossy-object-reconstruction-with-cost","title":"Glossy Object Reconstruction with Cost-effective Polarized Acquisition","date":"2025-04-09","arxiv_id":"2504.07025","repositories_listed":0,"syntology":null},{"url":null,"slug":"movsam-a-single-image-moving-object","title":"MovSAM: A Single-image Moving Object Segmentation Framework Based on Deep Thinking","date":"2025-04-09","arxiv_id":"2504.06863","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-self-supervised-framework-for-space-object","title":"A Self-Supervised Framework for Space Object Behaviour Characterisation","date":"2025-04-08","arxiv_id":"2504.06176","repositories_listed":0,"syntology":null},{"url":null,"slug":"d-feat-occlusions-diffusion-features-for","title":"D-Feat Occlusions: Diffusion Features for Robustness to Partial Visual Occlusions in Object Recognition","date":"2025-04-08","arxiv_id":"2504.06432","repositories_listed":0,"syntology":null},{"url":null,"slug":"primedrive-cot-a-precognitive-chain-of","title":"PRIMEDrive-CoT: A Precognitive Chain-of-Thought Framework for Uncertainty-Aware Object Interaction in Driving Scene Scenario","date":"2025-04-08","arxiv_id":"2504.05908","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounding-3d-object-affordance-with-language","title":"Grounding 3D Object Affordance with Language Instructions, Visual Observations and Interactions","date":"2025-04-07","arxiv_id":"2504.04744","repositories_listed":0,"syntology":null},{"url":null,"slug":"emf-event-meta-formers-for-event-based-real","title":"EMF: Event Meta Formers for Event-based Real-time Traffic Object Detection","date":"2025-04-05","arxiv_id":"2504.04124","repositories_listed":0,"syntology":null},{"url":null,"slug":"cornerpoint3d-look-at-the-nearest-corner","title":"CornerPoint3D: Look at the Nearest Corner Instead of the Center","date":"2025-04-03","arxiv_id":"2504.02464","repositories_listed":0,"syntology":null},{"url":null,"slug":"rasp-revisiting-3d-anamorphic-art-for-shadow","title":"RASP: Revisiting 3D Anamorphic Art for Shadow-Guided Packing of Irregular Objects","date":"2025-04-03","arxiv_id":"2504.02465","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-diffusion-based-framework-for-occluded","title":"A Diffusion-Based Framework for Occluded Object Movement","date":"2025-04-02","arxiv_id":"2504.01873","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-lg-track-an-enhanced-localization","title":"Deep LG-Track: An Enhanced Localization-Confidence-Guided Multi-Object Tracker","date":"2025-04-02","arxiv_id":"2504.01457","repositories_listed":0,"syntology":null},{"url":null,"slug":"slot-level-robotic-placement-via-visual","title":"Slot-Level Robotic Placement via Visual Imitation from Single Human Video","date":"2025-04-02","arxiv_id":"2504.01959","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-unified-referring-expression","title":"Towards Unified Referring Expression Segmentation Across Omni-Level Visual Target Granularities","date":"2025-04-02","arxiv_id":"2504.01954","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformerger-transformer-based-voice","title":"TransforMerger: Transformer-based Voice-Gesture Fusion for Robust Human-Robot Communication","date":"2025-04-02","arxiv_id":"2504.01708","repositories_listed":0,"syntology":null},{"url":null,"slug":"detail-aware-multi-view-stereo-network-for","title":"Detail-aware multi-view stereo network for depth estimation","date":"2025-03-31","arxiv_id":"2503.23684","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-isolated-attention-for-consistent","title":"Object Isolated Attention for Consistent Story Visualization","date":"2025-03-30","arxiv_id":"2503.23353","repositories_listed":0,"syntology":null},{"url":null,"slug":"physically-ground-commonsense-knowledge-for","title":"Physically Ground Commonsense Knowledge for Articulated Object Manipulation with Analytic Concepts","date":"2025-03-30","arxiv_id":"2503.23348","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-in-object-detection-a-systematic","title":"Context in object detection: a systematic literature review","date":"2025-03-29","arxiv_id":"2503.23249","repositories_listed":0,"syntology":null},{"url":null,"slug":"forcepose-a-deep-learning-approach-for-force","title":"ForcePose: A Deep Learning Approach for Force Calculation Based on Action Recognition Using MediaPipe Pose Estimation Combined with Object Detection","date":"2025-03-28","arxiv_id":"2503.22363","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperspectral-adapter-for-object-tracking","title":"Hyperspectral Adapter for Object Tracking based on Hyperspectral Video","date":"2025-03-28","arxiv_id":"2503.22199","repositories_listed":0,"syntology":null},{"url":null,"slug":"runa-object-level-out-of-distribution","title":"RUNA: Object-level Out-of-Distribution Detection via Regional Uncertainty Alignment of Multimodal Representations","date":"2025-03-28","arxiv_id":"2503.22285","repositories_listed":0,"syntology":null},{"url":null,"slug":"segment-then-splat-a-unified-approach-for-3d","title":"Segment then Splat: A Unified Approach for 3D Open-Vocabulary Segmentation based on Gaussian Splatting","date":"2025-03-28","arxiv_id":"2503.22204","repositories_listed":0,"syntology":null},{"url":null,"slug":"semalign3d-semantic-correspondence-between","title":"SemAlign3D: Semantic Correspondence between RGB-Images through Aligning 3D Object-Class Representations","date":"2025-03-28","arxiv_id":"2503.22462","repositories_listed":0,"syntology":null},{"url":null,"slug":"sight-single-image-conditioned-generation-of","title":"SIGHT: Single-Image Conditioned Generation of Hand Trajectories for Hand-Object Interaction","date":"2025-03-28","arxiv_id":"2503.22869","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-marine-debris-forward-looking-sonar","title":"The Marine Debris Forward-Looking Sonar Datasets","date":"2025-03-28","arxiv_id":"2503.22880","repositories_listed":0,"syntology":null},{"url":null,"slug":"transplat-lighting-consistent-cross-scene","title":"TranSplat: Lighting-Consistent Cross-Scene Object Transfer with 3D Gaussian Splatting","date":"2025-03-28","arxiv_id":"2503.22676","repositories_listed":0,"syntology":null},{"url":null,"slug":"vista-visual-contextual-and-text-augmented","title":"VisTa: Visual-contextual and Text-augmented Zero-shot Object-level OOD Detection","date":"2025-03-28","arxiv_id":"2503.22291","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctrl-o-language-controllable-object-centric","title":"CTRL-O: Language-Controllable Object-Centric Visual Representation Learning","date":"2025-03-27","arxiv_id":"2503.21747","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-jenga-discovering-object-dependencies","title":"Visual Jenga: Discovering Object Dependencies via Counterfactual Inpainting","date":"2025-03-27","arxiv_id":"2503.21770","repositories_listed":0,"syntology":null},{"url":null,"slug":"glrd-global-local-collaborative-reason-and","title":"GLRD: Global-Local Collaborative Reason and Debate with PSL for 3D Open-Vocabulary Detection","date":"2025-03-26","arxiv_id":"2503.20682","repositories_listed":0,"syntology":null},{"url":null,"slug":"guiding-human-object-interactions-with-rich","title":"Guiding Human-Object Interactions with Rich Geometry and Relations","date":"2025-03-26","arxiv_id":"2503.20172","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-object-keypoint-learning","title":"Incremental Object Keypoint Learning","date":"2025-03-26","arxiv_id":"2503.20248","repositories_listed":0,"syntology":null},{"url":null,"slug":"reltriple-learning-plausible-indoor-layouts","title":"RelTriple: Learning Plausible Indoor Layouts by Integrating Relationship Triples into the Diffusion Process","date":"2025-03-26","arxiv_id":"2503.20289","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-object-categories-multi-attribute","title":"Beyond Object Categories: Multi-Attribute Reference Understanding for Visual Grounding","date":"2025-03-25","arxiv_id":"2503.19240","repositories_listed":0,"syntology":null},{"url":null,"slug":"layercraft-enhancing-text-to-image-generation","title":"LayerCraft: Enhancing Text-to-Image Generation with CoT Reasoning and Layered Object Integration","date":"2025-03-25","arxiv_id":"2504.00010","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-3d-object-spatial-relationships-from","title":"Learning 3D Object Spatial Relationships from Pre-trained 2D Diffusion Models","date":"2025-03-25","arxiv_id":"2503.19914","repositories_listed":0,"syntology":null},{"url":null,"slug":"visuo-tactile-object-pose-estimation-for-a","title":"Visuo-Tactile Object Pose Estimation for a Multi-Finger Robot Hand with Low-Resolution In-Hand Tactile Sensing","date":"2025-03-25","arxiv_id":"2503.19893","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-human-object-interaction-synthesis","title":"Zero-Shot Human-Object Interaction Synthesis with Multimodal Priors","date":"2025-03-25","arxiv_id":"2503.20118","repositories_listed":0,"syntology":null}],"record_sha256":"e3edf9d782dcac1dbb037364942e40f46107ecee1602a52ed9ef0adaf8f19d13","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}