{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/object/papers/48","list_of":"/task/object","task":"Object","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":48,"pages_in_order":107,"rows_per_page":100,"rows":[4701,4800],"of":10696,"counts":{"archive_papers_tagged":10696,"with_a_code_link":3979,"where_syntology_ran_a_sample":1043,"not_listed_spam_title":0,"listed":10696,"listed_where_code_ran":1043,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":919,"every_run_a_failure_of_syntologys_instrument":124,"listed_with_a_run_with_no_instrument_failure":919,"listed_every_run_a_failure_of_syntologys_instrument":124,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/object","prev":"/task/object/papers/47","next":"/task/object/papers/49","papers":[{"url":null,"slug":"hand-object-interaction-pretraining-from","title":"Hand-Object Interaction Pretraining from Videos","date":"2024-09-12","arxiv_id":"2409.08273","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-object-event-graph-representation","title":"Multi-object event graph representation learning for Video Question Answering","date":"2024-09-12","arxiv_id":"2409.07747","repositories_listed":0,"syntology":null},{"url":null,"slug":"surgivid-annotation-efficient-surgical-video","title":"SURGIVID: Annotation-Efficient Surgical Video Object Discovery","date":"2024-09-12","arxiv_id":"2409.07801","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-view-3d-reconstruction-via-so-2","title":"Single-View 3D Reconstruction via SO(2)-Equivariant Gaussian Sculpting Networks","date":"2024-09-11","arxiv_id":"2409.07245","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bayesian-framework-for-active-object","title":"A Bayesian Framework for Active Tactile Object Recognition, Pose Estimation and Shape Transfer Learning","date":"2024-09-10","arxiv_id":"2409.06912","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-attribute-enriched-dataset-and-auto","title":"An Attribute-Enriched Dataset and Auto-Annotated Pipeline for Open Detection","date":"2024-09-10","arxiv_id":"2409.06300","repositories_listed":0,"syntology":null},{"url":null,"slug":"leia-latent-view-invariant-embeddings-for","title":"LEIA: Latent View-invariant Embeddings for Implicit 3D Articulation","date":"2024-09-10","arxiv_id":"2409.06703","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-modeling-from-underwater-forward-scan","title":"Object Modeling from Underwater Forward-Scan Sonar Imagery with Sea-Surface Multipath","date":"2024-09-10","arxiv_id":"2409.06815","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-words-to-poses-enhancing-novel-object","title":"From Words to Poses: Enhancing Novel Object Pose Estimation with Vision Language Models","date":"2024-09-09","arxiv_id":"2409.05413","repositories_listed":0,"syntology":null},{"url":null,"slug":"lsvos-challenge-report-large-scale-complex","title":"LSVOS Challenge Report: Large-scale Complex and Long Video Object Segmentation","date":"2024-09-09","arxiv_id":"2409.05847","repositories_listed":0,"syntology":null},{"url":null,"slug":"replay-consolidation-with-label-propagation","title":"Replay Consolidation with Label Propagation for Continual Object Detection","date":"2024-09-09","arxiv_id":"2409.05650","repositories_listed":0,"syntology":null},{"url":"/paper/rcbevdet-toward-high-accuracy-radar-camera","slug":"rcbevdet-toward-high-accuracy-radar-camera","title":"RCBEVDet++: Toward High-accuracy Radar-Camera Fusion 3D Perception Network","date":"2024-09-08","arxiv_id":"2409.04979","repositories_listed":0,"syntology":null},{"url":null,"slug":"dense-hand-object-ho-graspnet-with-full","title":"Dense Hand-Object(HO) GraspNet with Full Grasping Taxonomy and Dynamics","date":"2024-09-06","arxiv_id":"2409.04033","repositories_listed":0,"syntology":null},{"url":null,"slug":"future-does-matter-boosting-3d-object","title":"Future Does Matter: Boosting 3D Object Detection with Temporal Motion Estimation in Point Cloud Sequences","date":"2024-09-06","arxiv_id":"2409.04390","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-diffusion-for-hand-object-grasp","title":"Multi-Modal Diffusion for Hand-Object Grasp Generation","date":"2024-09-06","arxiv_id":"2409.04560","repositories_listed":0,"syntology":null},{"url":null,"slug":"thinking-outside-the-bbox-unconstrained","title":"Thinking Outside the BBox: Unconstrained Generative Object Compositing","date":"2024-09-06","arxiv_id":"2409.04559","repositories_listed":0,"syntology":null},{"url":null,"slug":"organized-grouped-discrete-representation-for","title":"Organized Grouped Discrete Representation for Object-Centric Learning","date":"2024-09-05","arxiv_id":"2409.03553","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-gaussian-for-monocular-6d-pose","title":"Object Gaussian for Monocular 6D Pose Estimation from Sparse Views","date":"2024-09-04","arxiv_id":"2409.02581","repositories_listed":0,"syntology":null},{"url":null,"slug":"pluralistic-salient-object-detection","title":"Pluralistic Salient Object Detection","date":"2024-09-04","arxiv_id":"2409.02368","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-modern-take-on-visual-relationship","title":"A Modern Take on Visual Relationship Reasoning for Grasp Planning","date":"2024-09-03","arxiv_id":"2409.02035","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-apple-object-detection-with","title":"Improving Apple Object Detection with Occlusion-Enhanced Distillation","date":"2024-09-03","arxiv_id":"2409.01573","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-allocations-with-capacity-constrained","title":"Optimal allocations with capacity constrained verification","date":"2024-09-03","arxiv_id":"2409.02031","repositories_listed":0,"syntology":null},{"url":null,"slug":"ds-myolo-a-reliable-object-detector-based-on","title":"DS MYOLO: A Reliable Object Detector Based on SSMs for Driving Scenarios","date":"2024-09-02","arxiv_id":"2409.01093","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-pixels-to-objects-a-hierarchical","title":"From Pixels to Objects: A Hierarchical Approach for Part and Object Segmentation Using Local and Global Aggregation","date":"2024-09-02","arxiv_id":"2409.01353","repositories_listed":0,"syntology":null},{"url":null,"slug":"comogen-a-controllable-text-to-3d-multi","title":"COMOGen: A Controllable Text-to-3D Multi-object Generation Framework","date":"2024-09-01","arxiv_id":"2409.00590","repositories_listed":0,"syntology":null},{"url":null,"slug":"detection-recognition-and-pose-estimation-of","title":"Detection, Recognition and Pose Estimation of Tabletop Objects","date":"2024-09-01","arxiv_id":"2409.00869","repositories_listed":0,"syntology":null},{"url":null,"slug":"remove-a-reference-free-metric-for-object","title":"ReMOVE: A Reference-free Metric for Object Erasure","date":"2024-09-01","arxiv_id":"2409.00707","repositories_listed":0,"syntology":null},{"url":null,"slug":"erasedraw-learning-to-insert-objects-by","title":"EraseDraw: Learning to Draw Step-by-Step via Erasing Objects from Images","date":"2024-08-31","arxiv_id":"2409.00522","repositories_listed":0,"syntology":null},{"url":null,"slug":"bop-d-revisiting-6d-pose-estimation-benchmark","title":"BOP-Distrib: Revisiting 6D Pose Estimation Benchmarks for Better Evaluation under Visual Ambiguities","date":"2024-08-30","arxiv_id":"2408.17297","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-errors-in-controlled-turret-system","title":"Analyzing Errors in Controlled Turret System Given Target Location Input from Artificial Intelligence Methods in Automatic Target Recognition","date":"2024-08-29","arxiv_id":"2408.16923","repositories_listed":0,"syntology":null},{"url":null,"slug":"anno-incomplete-multi-dataset-detection","title":"Anno-incomplete Multi-dataset Detection","date":"2024-08-29","arxiv_id":"2408.16247","repositories_listed":0,"syntology":null},{"url":null,"slug":"discriminative-spatial-semantic-vos-solution","title":"Discriminative Spatial-Semantic VOS Solution: 1st Place Solution for 6th LSVOS","date":"2024-08-29","arxiv_id":"2408.16431","repositories_listed":0,"syntology":null},{"url":"/paper/partformer-awakening-latent-diverse","slug":"partformer-awakening-latent-diverse","title":"PartFormer: Awakening Latent Diverse Representation from Vision Transformer for Object Re-Identification","date":"2024-08-29","arxiv_id":"2408.16684","repositories_listed":0,"syntology":null},{"url":null,"slug":"microyolo-towards-single-shot-object","title":"microYOLO: Towards Single-Shot Object Detection on Microcontrollers","date":"2024-08-28","arxiv_id":"2408.15865","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-detection-for-vehicle-dashcams-using","title":"Object Detection for Vehicle Dashcams using Transformers","date":"2024-08-28","arxiv_id":"2408.15809","repositories_listed":0,"syntology":null},{"url":null,"slug":"small-object-detection-for-indoor-assistance","title":"Small Object Detection for Indoor Assistance to the Blind using YOLO NAS Small and Super Gradients","date":"2024-08-28","arxiv_id":"2409.07469","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-is-yolov8-an-in-depth-exploration-of-the","title":"What is YOLOv8: An In-Depth Exploration of the Internal Features of the Next-Generation Object Detector","date":"2024-08-28","arxiv_id":"2408.15857","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-on-the-position-encoding-in","title":"An Investigation on The Position Encoding in Vision-Based Dynamics Prediction","date":"2024-08-27","arxiv_id":"2408.15201","repositories_listed":0,"syntology":null},{"url":null,"slug":"build-a-scene-interactive-3d-layout-control","title":"Build-A-Scene: Interactive 3D Layout Control for Diffusion-Based Image Generation","date":"2024-08-27","arxiv_id":"2408.14819","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-aware-manipulation-with-object-centric","title":"3D-Aware Manipulation with Object-Centric Gaussian Splatting","date":"2024-08-26","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-few-shot-object-detection-a-detailed","title":"Beyond Few-shot Object Detection: A Detailed Survey","date":"2024-08-26","arxiv_id":"2408.14249","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-local-pattern-modularization-for","title":"Learning Local Pattern Modularization for Point Cloud Reconstruction from Unseen Classes","date":"2024-08-26","arxiv_id":"2408.14279","repositories_listed":0,"syntology":null},{"url":null,"slug":"pvafn-point-voxel-attention-fusion-network","title":"PVAFN: Point-Voxel Attention Fusion Network with Multi-Pooling Enhancing for 3D Object Detection","date":"2024-08-26","arxiv_id":"2408.14600","repositories_listed":0,"syntology":null},{"url":null,"slug":"intertrack-tracking-human-object-interaction","title":"InterTrack: Tracking Human Object Interaction without Object Templates","date":"2024-08-25","arxiv_id":"2408.13953","repositories_listed":0,"syntology":null},{"url":null,"slug":"css-segment-2nd-place-report-of-lsvos","title":"CSS-Segment: 2nd Place Report of LSVOS Challenge VOS Track","date":"2024-08-24","arxiv_id":"2408.13582","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralised-gradient-based-variational","title":"Decentralised Variational Inference Frameworks for Multi-object Tracking on Sensor Networks: Additional Notes","date":"2024-08-24","arxiv_id":"2408.13689","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-temporal-embedding-of-objects","title":"Context-Aware Temporal Embedding of Objects in Video Data","date":"2024-08-23","arxiv_id":"2408.12789","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-crucial-objects-in-blind-and-low","title":"Identifying Crucial Objects in Blind and Low-Vision Individuals' Navigation","date":"2024-08-23","arxiv_id":"2408.13175","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-2d-invariant-affordance-knowledge","title":"Learning 2D Invariant Affordance Knowledge for 3D Affordance Grounding","date":"2024-08-23","arxiv_id":"2408.13024","repositories_listed":0,"syntology":null},{"url":null,"slug":"mctr-multi-camera-tracking-transformer","title":"MCTR: Multi Camera Tracking Transformer","date":"2024-08-23","arxiv_id":"2408.13243","repositories_listed":0,"syntology":null},{"url":null,"slug":"shapeicp-iterative-category-level-object-pose","title":"ShapeICP: Iterative Category-level Object Pose and Shape Estimation from Depth","date":"2024-08-23","arxiv_id":"2408.13147","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-learning-digital-twin-case-study-on","title":"Towards learning digital twin: case study on an anisotropic non-ideal rotor system","date":"2024-08-23","arxiv_id":"2408.13021","repositories_listed":0,"syntology":null},{"url":null,"slug":"banktweak-adversarial-attack-against-multi","title":"BankTweak: Adversarial Attack against Multi-Object Trackers by Manipulating Feature Banks","date":"2024-08-22","arxiv_id":"2408.12727","repositories_listed":0,"syntology":null},{"url":null,"slug":"catfree3d-category-agnostic-3d-object","title":"CatFree3D: Category-agnostic 3D Object Detection with Diffusion","date":"2024-08-22","arxiv_id":"2408.12747","repositories_listed":0,"syntology":null},{"url":null,"slug":"class-balanced-open-set-semi-supervised","title":"Class-balanced Open-set Semi-supervised Object Detection for Medical Images","date":"2024-08-22","arxiv_id":"2408.12355","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-invariant-progressive-knowledge","title":"Domain-invariant Progressive Knowledge Distillation for UAV-based Object Detection","date":"2024-08-21","arxiv_id":"2408.11407","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-object-count-optimization-for-text","title":"Detection-Driven Object Count Optimization for Text-to-Image Diffusion Models","date":"2024-08-21","arxiv_id":"2408.11721","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-incremental-explanations-for-object","title":"Real-Time Incremental Explanations for Object Detectors","date":"2024-08-21","arxiv_id":"2408.11963","repositories_listed":0,"syntology":null},{"url":null,"slug":"sbdet-a-symmetry-breaking-object-detector-via","title":"SBDet: A Symmetry-Breaking Object Detector via Relaxed Rotation-Equivariance","date":"2024-08-21","arxiv_id":"2408.11760","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-human-object-interaction","title":"A Review of Human-Object Interaction Detection","date":"2024-08-20","arxiv_id":"2408.10641","repositories_listed":0,"syntology":null},{"url":null,"slug":"just-a-hint-point-supervised-camouflaged","title":"Just a Hint: Point-Supervised Camouflaged Object Detection","date":"2024-08-20","arxiv_id":"2408.10777","repositories_listed":0,"syntology":null},{"url":null,"slug":"lsvos-challenge-3rd-place-report-sam2-and","title":"LSVOS Challenge 3rd Place Report: SAM2 and Cutie based VOS","date":"2024-08-20","arxiv_id":"2408.10469","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-potential-of-open-vocabulary-models","title":"On the Potential of Open-Vocabulary Models for Object Detection in Unusual Street Scenes","date":"2024-08-20","arxiv_id":"2408.11221","repositories_listed":0,"syntology":null},{"url":null,"slug":"target-oriented-object-grasping-via","title":"Target-Oriented Object Grasping via Multimodal Human Guidance","date":"2024-08-20","arxiv_id":"2408.11138","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-aware-instance-segmentation-and-tracking","title":"3D-Aware Instance Segmentation and Tracking in Egocentric Videos","date":"2024-08-19","arxiv_id":"2408.09860","repositories_listed":0,"syntology":null},{"url":null,"slug":"disconerf-class-agnostic-object-field-for-3d","title":"Enforcing View-Consistency in Class-Agnostic 3D Segmentation Fields","date":"2024-08-19","arxiv_id":"2408.09928","repositories_listed":0,"syntology":null},{"url":null,"slug":"photorealistic-object-insertion-with","title":"Photorealistic Object Insertion with Diffusion-Guided Inverse Rendering","date":"2024-08-19","arxiv_id":"2408.09702","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-object-segmentation-via-sam-2-the-4th","title":"Video Object Segmentation via SAM 2: The 4th Solution for LSVOS Challenge VOS Track","date":"2024-08-19","arxiv_id":"2408.10125","repositories_listed":0,"syntology":null},{"url":null,"slug":"retina-inspired-object-motion-segmentation","title":"Retina-Inspired Object Motion Segmentation for Event-Cameras","date":"2024-08-18","arxiv_id":"2408.09454","repositories_listed":0,"syntology":null},{"url":null,"slug":"depth-guided-texture-diffusion-for-image","title":"Depth-guided Texture Diffusion for Image Semantic Segmentation","date":"2024-08-17","arxiv_id":"2408.09097","repositories_listed":0,"syntology":null},{"url":null,"slug":"gslamot-a-tracklet-and-query-graph-based","title":"GSLAMOT: A Tracklet and Query Graph-based Simultaneous Locating, Mapping, and Multiple Object Tracking System","date":"2024-08-17","arxiv_id":"2408.09191","repositories_listed":0,"syntology":null},{"url":null,"slug":"maskbev-towards-a-unified-framework-for-bev","title":"MaskBEV: Towards A Unified Framework for BEV Detection and Map Segmentation","date":"2024-08-17","arxiv_id":"2408.09122","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-object-centric-representation","title":"Zero-Shot Object-Centric Representation Learning","date":"2024-08-17","arxiv_id":"2408.09162","repositories_listed":0,"syntology":null},{"url":null,"slug":"achieving-complex-image-edits-via-function","title":"FunEditor: Achieving Complex Image Edits via Function Aggregation with Diffusion Models","date":"2024-08-16","arxiv_id":"2408.08495","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-object-detection-with-hybrid","title":"Enhancing Object Detection with Hybrid dataset in Manufacturing Environments: Comparing Federated Learning to Conventional Techniques","date":"2024-08-16","arxiv_id":"2408.08974","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-relational-triple-extraction-with","title":"Multimodal Relational Triple Extraction with Query-based Entity Object Transformer","date":"2024-08-16","arxiv_id":"2408.08709","repositories_listed":0,"syntology":null},{"url":null,"slug":"textoc-text-driven-object-centric-style","title":"TEXTOC: Text-driven Object-Centric Style Transfer","date":"2024-08-16","arxiv_id":"2408.08461","repositories_listed":0,"syntology":null},{"url":null,"slug":"infra-yolo-efficient-neural-network-structure","title":"Infra-YOLO: Efficient Neural Network Structure with Model Compression for Real-Time Infrared Small Object Detection","date":"2024-08-14","arxiv_id":"2408.07455","repositories_listed":0,"syntology":null},{"url":"/paper/see-it-all-contextualized-late-aggregation","slug":"see-it-all-contextualized-late-aggregation","title":"See It All: Contextualized Late Aggregation for 3D Dense Captioning","date":"2024-08-14","arxiv_id":"2408.07648","repositories_listed":0,"syntology":null},{"url":"/paper/bi-directional-contextual-attention-for-3d","slug":"bi-directional-contextual-attention-for-3d","title":"Bi-directional Contextual Attention for 3D Dense Captioning","date":"2024-08-13","arxiv_id":"2408.06662","repositories_listed":0,"syntology":null},{"url":null,"slug":"divide-and-conquer-improving-multi-camera-3d","title":"Divide and Conquer: Improving Multi-Camera 3D Perception with 2D Semantic-Depth Priors and Input-Dependent Queries","date":"2024-08-13","arxiv_id":"2408.06901","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-domain-shift-on-radar-based-3d","title":"Exploring Domain Shift on Radar-Based 3D Object Detection Amidst Diverse Environmental Conditions","date":"2024-08-13","arxiv_id":"2408.06772","repositories_listed":0,"syntology":null},{"url":null,"slug":"scenegpt-a-language-model-for-3d-scene","title":"SceneGPT: A Language Model for 3D Scene Understanding","date":"2024-08-13","arxiv_id":"2408.06926","repositories_listed":0,"syntology":null},{"url":null,"slug":"slotlifter-slot-guided-feature-lifting-for","title":"SlotLifter: Slot-guided Feature Lifting for Learning Object-centric Radiance Fields","date":"2024-08-13","arxiv_id":"2408.06697","repositories_listed":0,"syntology":null},{"url":null,"slug":"dpdetr-decoupled-position-detection","title":"DPDETR: Decoupled Position Detection Transformer for Infrared-Visible Object Detection","date":"2024-08-12","arxiv_id":"2408.06123","repositories_listed":0,"syntology":null},{"url":null,"slug":"mv2dfusion-leveraging-modality-specific","title":"MV2DFusion: Leveraging Modality-Specific Object Semantics for Multi-Modal 3D Detection","date":"2024-08-12","arxiv_id":"2408.05945","repositories_listed":0,"syntology":null},{"url":null,"slug":"macformer-semantic-segmentation-with-fine","title":"MacFormer: Semantic Segmentation with Fine Object Boundaries","date":"2024-08-11","arxiv_id":"2408.05699","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-domain-generalization-for-multi-modal","title":"Robust Domain Generalization for Multi-modal Object Recognition","date":"2024-08-11","arxiv_id":"2408.05831","repositories_listed":0,"syntology":null},{"url":null,"slug":"saber-6d-shape-representation-based-implicit","title":"SABER-6D: Shape Representation Based Implicit Object Pose Estimation","date":"2024-08-11","arxiv_id":"2408.05867","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-uncertainty-aware-object","title":"Embodied Uncertainty-Aware Object Segmentation","date":"2024-08-08","arxiv_id":"2408.04760","repositories_listed":0,"syntology":null},{"url":"/paper/img-diff-contrastive-data-synthesis-for","slug":"img-diff-contrastive-data-synthesis-for","title":"Img-Diff: Contrastive Data Synthesis for Multimodal Large Language Models","date":"2024-08-08","arxiv_id":"2408.04594","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-03178","title":"An Object is Worth 64x64 Pixels: Generating 3D Object via Image Diffusion","date":"2024-08-06","arxiv_id":"2408.03178","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-03238","title":"LAC-Net: Linear-Fusion Attention-Guided Convolutional Network for Accurate Robotic Grasping Under the Occlusion","date":"2024-08-06","arxiv_id":"2408.03238","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-hallucinations-in-large-vision-1","title":"Mitigating Hallucinations in Large Vision-Language Models (LVLMs) via Language-Contrastive Decoding (LCD)","date":"2024-08-06","arxiv_id":"2408.04664","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-how-blind-users-handle-object","title":"Understanding How Blind Users Handle Object Recognition Errors: Strategies and Challenges","date":"2024-08-06","arxiv_id":"2408.03303","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01934","title":"A Survey and Evaluation of Adversarial Attacks for Object Detection","date":"2024-08-04","arxiv_id":"2408.01934","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02088","title":"KAN-RCBEVDepth: A multi-modal fusion algorithm in object detection for autonomous driving","date":"2024-08-04","arxiv_id":"2408.02088","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02100","title":"View-consistent Object Removal in Radiance Fields","date":"2024-08-04","arxiv_id":"2408.02100","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01655","title":"Stimulating Imagination: Towards General-purpose Object Rearrangement","date":"2024-08-03","arxiv_id":"2408.01655","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01739","title":"LAM3D: Leveraging Attention for Monocular 3D Object Detection","date":"2024-08-03","arxiv_id":"2408.01739","repositories_listed":0,"syntology":null}],"record_sha256":"649a4bf494cdd5d4fb5d882144fff95b8f30a53006d65d2a1465b496101c270c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}