{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/object-detection/papers/53","list_of":"/task/object-detection","task":"Object Detection","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":53,"pages_in_order":110,"rows_per_page":100,"rows":[5201,5300],"of":10957,"counts":{"archive_papers_tagged":10957,"with_a_code_link":4657,"where_syntology_ran_a_sample":1183,"not_listed_spam_title":0,"listed":10957,"listed_where_code_ran":1183,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1038,"every_run_a_failure_of_syntologys_instrument":145,"listed_with_a_run_with_no_instrument_failure":1038,"listed_every_run_a_failure_of_syntologys_instrument":145,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/object-detection","prev":"/task/object-detection/papers/52","next":"/task/object-detection/papers/54","papers":[{"url":null,"slug":"assessing-the-performance-of-ct-image","title":"Assessing the performance of CT image denoisers using Laguerre-Gauss Channelized Hotelling Observer for lesion detection","date":"2024-12-04","arxiv_id":"2412.02920","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-fusion-of-semantic-and-depth-information","title":"Data Fusion of Semantic and Depth Information in the Context of Object Detection","date":"2024-12-04","arxiv_id":"2412.03490","repositories_listed":0,"syntology":null},{"url":null,"slug":"nbm-an-open-dataset-for-the-acoustic","title":"NBM: an Open Dataset for the Acoustic Monitoring of Nocturnal Migratory Birds in Europe","date":"2024-12-04","arxiv_id":"2412.03633","repositories_listed":0,"syntology":null},{"url":null,"slug":"objectfinder-open-vocabulary-assistive-system","title":"ObjectFinder: An Open-Vocabulary Assistive System for Interactive Object Search by Blind People","date":"2024-12-04","arxiv_id":"2412.03118","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-tokens-enhance-visual-reasoning-in","title":"Perception Tokens Enhance Visual Reasoning in Multimodal Language Models","date":"2024-12-04","arxiv_id":"2412.03548","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-driven-image-fusion-with-learnable","title":"Task-driven Image Fusion with Learnable Fusion Loss","date":"2024-12-04","arxiv_id":"2412.03240","repositories_listed":0,"syntology":null},{"url":null,"slug":"trend-unsupervised-3d-representation-learning","title":"TREND: Unsupervised 3D Representation Learning via Temporal Forecasting for LiDAR Perception","date":"2024-12-04","arxiv_id":"2412.03054","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimized-cnns-for-rapid-3d-point-cloud","title":"Optimized CNNs for Rapid 3D Point Cloud Object Recognition","date":"2024-12-03","arxiv_id":"2412.02855","repositories_listed":0,"syntology":null},{"url":null,"slug":"redundant-queries-in-detr-based-3d-detection","title":"Redundant Queries in DETR-Based 3D Detection Methods: Unnecessary and Prunable","date":"2024-12-03","arxiv_id":"2412.02054","repositories_listed":0,"syntology":null},{"url":null,"slug":"underload-defending-against-latency-attacks","title":"Can't Slow me Down: Learning Robust and Hardware-Adaptive Object Detectors against Latency Attacks for Edge Devices","date":"2024-12-03","arxiv_id":"2412.02171","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-instance-navigation-leveraging","title":"Collaborative Instance Navigation: Leveraging Agent Self-Dialogue to Minimize User Input","date":"2024-12-02","arxiv_id":"2412.01250","repositories_listed":0,"syntology":null},{"url":null,"slug":"divide-and-conquer-confluent-triple-flow","title":"Divide-and-Conquer: Confluent Triple-Flow Network for RGB-T Salient Object Detection","date":"2024-12-02","arxiv_id":"2412.01556","repositories_listed":0,"syntology":null},{"url":null,"slug":"gfreedet-exploiting-gaussian-splatting-and","title":"GFreeDet: Exploiting Gaussian Splatting and Foundation Models for Model-free Unseen Object Detection in the BOP Challenge 2024","date":"2024-12-02","arxiv_id":"2412.01552","repositories_listed":0,"syntology":null},{"url":null,"slug":"hprm-high-performance-robotic-middleware-for","title":"HPRM: High-Performance Robotic Middleware for Intelligent Autonomous Systems","date":"2024-12-02","arxiv_id":"2412.01799","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-object-detection-by-modifying","title":"Improving Object Detection by Modifying Synthetic Data with Explainable AI","date":"2024-12-02","arxiv_id":"2412.01477","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-parking-with-pixel-wise-roi-selection","title":"Smart Parking with Pixel-Wise ROI Selection for Vehicle Detection Using YOLOv8, YOLOv9, YOLOv10, and YOLOv11","date":"2024-12-02","arxiv_id":"2412.01983","repositories_listed":0,"syntology":null},{"url":"/paper/bev-sushi-multi-target-multi-camera-3d","slug":"bev-sushi-multi-target-multi-camera-3d","title":"MCBLT: Multi-Camera Multi-Object 3D Tracking in Long Videos","date":"2024-12-01","arxiv_id":"2412.00692","repositories_listed":0,"syntology":null},{"url":null,"slug":"sketch-guided-motion-diffusion-for-stylized","title":"Sketch-Guided Motion Diffusion for Stylized Cinemagraph Synthesis","date":"2024-12-01","arxiv_id":"2412.00638","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-technologies-with-applications-in","title":"Vision Technologies with Applications in Traffic Surveillance Systems: A Holistic Survey","date":"2024-11-30","arxiv_id":"2412.00348","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-anomaly-detection-in-video-streams","title":"Real-Time Anomaly Detection in Video Streams","date":"2024-11-29","arxiv_id":"2411.19731","repositories_listed":0,"syntology":null},{"url":"/paper/sparc-sparse-radar-camera-fusion-for-3d","slug":"sparc-sparse-radar-camera-fusion-for-3d","title":"SpaRC: Sparse Radar-Camera Fusion for 3D Object Detection","date":"2024-11-29","arxiv_id":"2411.19860","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-prompt-generation-and-grounding","title":"Automatic Prompt Generation and Grounding Object Detection for Zero-Shot Image Anomaly Detection","date":"2024-11-28","arxiv_id":"2411.19220","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-learning-towards-semi-supervised-object","title":"Co-Learning: Towards Semi-Supervised Object Detection with Road-side Cameras","date":"2024-11-28","arxiv_id":"2411.19143","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-performance-evaluation-of-1","title":"Comprehensive Performance Evaluation of YOLOv11, YOLOv10, YOLOv9, YOLOv8 and YOLOv5 on Object Detection of Power Equipment","date":"2024-11-28","arxiv_id":"2411.18871","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-attention-and-bi-directional-fusion","title":"Dynamic Attention and Bi-directional Fusion for Safety Helmet Wearing Detection","date":"2024-11-28","arxiv_id":"2411.19071","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-batch-normalization-with-tta-for","title":"Improving Batch Normalization with TTA for Robust Object Detection in Self-Driving","date":"2024-11-28","arxiv_id":"2411.18860","repositories_listed":0,"syntology":null},{"url":null,"slug":"mvformer-diversifying-feature-normalization","title":"MVFormer: Diversifying Feature Normalization and Token Mixing for Efficient Vision Transformers","date":"2024-11-28","arxiv_id":"2411.18995","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-moving-object-segmentation-from-monocular","title":"On Moving Object Segmentation from Monocular Video with Transformers","date":"2024-11-28","arxiv_id":"2411.19141","repositories_listed":0,"syntology":null},{"url":null,"slug":"hdi-former-hybrid-dynamic-interaction-ann-snn","title":"HDI-Former: Hybrid Dynamic Interaction ANN-SNN Transformer for Object Detection Using Frames and Events","date":"2024-11-27","arxiv_id":"2411.18658","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-multispectral-object-detection-a","title":"Optimizing Multispectral Object Detection: A Bag of Tricks and Comprehensive Benchmarks","date":"2024-11-27","arxiv_id":"2411.18288","repositories_listed":0,"syntology":null},{"url":null,"slug":"roictrl-boosting-instance-control-for-visual","title":"ROICtrl: Boosting Instance Control for Visual Generation","date":"2024-11-27","arxiv_id":"2411.17949","repositories_listed":0,"syntology":null},{"url":null,"slug":"rpee-heads-a-novel-benchmark-for-pedestrian","title":"RPEE-HEADS: A Novel Benchmark for Pedestrian Head Detection in Crowd Videos","date":"2024-11-27","arxiv_id":"2411.18164","repositories_listed":0,"syntology":null},{"url":null,"slug":"dgnn-yolo-dynamic-graph-neural-networks-with","title":"Interpretable Dynamic Graph Neural Networks for Small Occluded Object Detection and Tracking","date":"2024-11-26","arxiv_id":"2411.17251","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-aleatoric-uncertainty-in-object","title":"Exploring Aleatoric Uncertainty in Object Detection via Vision Foundation Models","date":"2024-11-26","arxiv_id":"2411.17767","repositories_listed":0,"syntology":null},{"url":"/paper/cuts3d-cutting-semantics-in-3d-for-2d","slug":"cuts3d-cutting-semantics-in-3d-for-2d","title":"CutS3D: Cutting Semantics in 3D for 2D Unsupervised Instance Segmentation","date":"2024-11-25","arxiv_id":"2411.16319","repositories_listed":0,"syntology":null},{"url":null,"slug":"diagnosis-of-diabetic-retinopathy-using","title":"Diagnosis of diabetic retinopathy using machine learning & deep learning technique","date":"2024-11-25","arxiv_id":"2411.16250","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperspectral-image-cross-domain-object","title":"Hyperspectral Image Cross-Domain Object Detection Method based on Spectral-Spatial Feature Alignment","date":"2024-11-25","arxiv_id":"2411.16772","repositories_listed":0,"syntology":null},{"url":null,"slug":"imperceptible-adversarial-examples-in-the","title":"Imperceptible Adversarial Examples in the Physical World","date":"2024-11-25","arxiv_id":"2411.16622","repositories_listed":0,"syntology":null},{"url":null,"slug":"leverage-task-context-for-object-affordance","title":"Leverage Task Context for Object Affordance Ranking","date":"2024-11-25","arxiv_id":"2411.16082","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-episodic-memory-visual-query","title":"Online Episodic Memory Visual Query Localization with Egocentric Streaming Object Memory","date":"2024-11-25","arxiv_id":"2411.16934","repositories_listed":0,"syntology":null},{"url":null,"slug":"anysynth-harnessing-the-power-of-image","title":"AnySynth: Harnessing the Power of Image Synthetic Data Generation for Generalized Vision-Language Tasks","date":"2024-11-24","arxiv_id":"2411.16749","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-object-detection-accuracy-in","title":"Enhancing Object Detection Accuracy in Autonomous Vehicles Using Synthetic Data","date":"2024-11-23","arxiv_id":"2411.15602","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-open-vocabulary-object","title":"Fine-Grained Open-Vocabulary Object Recognition via User-Guided Segmentation","date":"2024-11-23","arxiv_id":"2411.15620","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-an-open-vocabulary-monocular-3d","title":"Training an Open-Vocabulary Monocular 3D Object Detection Model without 3D Data","date":"2024-11-23","arxiv_id":"2411.15657","repositories_listed":0,"syntology":null},{"url":null,"slug":"twin-trigger-generative-networks-for-backdoor","title":"Twin Trigger Generative Networks for Backdoor Attacks against Object Detection","date":"2024-11-23","arxiv_id":"2411.15439","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-real-time-detr-approach-to-bangladesh-road","title":"A Real-Time DETR Approach to Bangladesh Road Object Detection for Autonomous Vehicles","date":"2024-11-22","arxiv_id":"2411.15110","repositories_listed":0,"syntology":null},{"url":null,"slug":"mssf-a-4d-radar-and-camera-fusion-framework","title":"MSSF: A 4D Radar and Camera Fusion Framework With Multi-Stage Sampling for 3D Object Detection in Autonomous Driving","date":"2024-11-22","arxiv_id":"2411.15016","repositories_listed":0,"syntology":null},{"url":null,"slug":"visionpad-a-vision-centric-pre-training","title":"VisionPAD: A Vision-Centric Pre-training Paradigm for Autonomous Driving","date":"2024-11-22","arxiv_id":"2411.14716","repositories_listed":0,"syntology":null},{"url":null,"slug":"multitask-learning-for-sar-ship-detection","title":"Multitask Learning for SAR Ship Detection with Gaussian-Mask Joint Segmentation","date":"2024-11-21","arxiv_id":"2411.13847","repositories_listed":0,"syntology":null},{"url":"/paper/transforming-static-images-using-generative","slug":"transforming-static-images-using-generative","title":"Transforming Static Images Using Generative Models for Video Salient Object Detection","date":"2024-11-21","arxiv_id":"2411.13975","repositories_listed":0,"syntology":null},{"url":null,"slug":"bounding-box-watermarking-defense-against","title":"Bounding-box Watermarking: Defense against Model Extraction Attacks on Object Detectors","date":"2024-11-20","arxiv_id":"2411.13047","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-feature-logits-contrastive","title":"Collaborative Feature-Logits Contrastive Learning for Open-Set Semi-Supervised Object Detection","date":"2024-11-20","arxiv_id":"2411.13001","repositories_listed":0,"syntology":null},{"url":null,"slug":"dis-mine-instance-segmentation-for-disaster","title":"DIS-Mine: Instance Segmentation for Disaster-Awareness in Poor-Light Condition in Underground Mines","date":"2024-11-20","arxiv_id":"2411.13544","repositories_listed":0,"syntology":null},{"url":null,"slug":"mambadetr-query-based-temporal-modeling-using","title":"MambaDETR: Query-based Temporal Modeling using State Space Model for Multi-View 3D Object Detection","date":"2024-11-20","arxiv_id":"2411.13628","repositories_listed":0,"syntology":null},{"url":null,"slug":"vadet-multi-frame-lidar-3d-object-detection","title":"VADet: Multi-frame LiDAR 3D Object Detection using Variable Aggregation","date":"2024-11-20","arxiv_id":"2411.13186","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-emerging-trends-and-research","title":"Exploring Emerging Trends and Research Opportunities in Visual Place Recognition","date":"2024-11-18","arxiv_id":"2411.11481","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-deep-learning-research-with","title":"Scaling Deep Learning Research with Kubernetes on the NRP Nautilus HyperCluster","date":"2024-11-18","arxiv_id":"2411.12038","repositories_listed":0,"syntology":null},{"url":null,"slug":"sl-yolo-a-stronger-and-lighter-drone-target","title":"SL-YOLO: A Stronger and Lighter Drone Target Detection Model","date":"2024-11-18","arxiv_id":"2411.11477","repositories_listed":0,"syntology":null},{"url":null,"slug":"woodyolo-a-novel-object-detector-for-wood","title":"WoodYOLO: A Novel Object Detector for Wood Species Detection in Microscopic Images","date":"2024-11-18","arxiv_id":"2411.11738","repositories_listed":0,"syntology":null},{"url":null,"slug":"evt-efficient-view-transformation-for-multi","title":"EVT: Efficient View Transformation for Multi-Modal 3D Object Detection","date":"2024-11-16","arxiv_id":"2411.10715","repositories_listed":0,"syntology":null},{"url":null,"slug":"diachronic-document-dataset-for-semantic","title":"Diachronic Document Dataset for Semantic Layout Analysis","date":"2024-11-15","arxiv_id":"2411.10068","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-image-based-aphid-counting-in","title":"Interactive Image-Based Aphid Counting in Yellow Water Traps under Stirring Actions","date":"2024-11-15","arxiv_id":"2411.10357","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-ai-driven-people-tracking-and","title":"Real-Time AI-Driven People Tracking and Counting Using Overhead Cameras","date":"2024-11-15","arxiv_id":"2411.10072","repositories_listed":0,"syntology":null},{"url":null,"slug":"structure-tensor-representation-for-robust","title":"Structure Tensor Representation for Robust Oriented Object Detection","date":"2024-11-15","arxiv_id":"2411.10497","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-linguistic-agent-towards-collaborative","title":"Visual-Linguistic Agent: Towards Collaborative Contextual Object Reasoning","date":"2024-11-15","arxiv_id":"2411.10252","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-attacks-using-differentiable","title":"RenderBender: A Survey on Adversarial Attacks Using Differentiable Rendering","date":"2024-11-14","arxiv_id":"2411.09749","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-consistency-in-multimodal-large","title":"Cross-Modal Consistency in Multimodal Large Language Models","date":"2024-11-14","arxiv_id":"2411.09273","repositories_listed":0,"syntology":null},{"url":null,"slug":"dt-jrd-deep-transformer-based-just","title":"DT-JRD: Deep Transformer based Just Recognizable Difference Prediction Model for Video Coding for Machines","date":"2024-11-14","arxiv_id":"2411.09308","repositories_listed":0,"syntology":null},{"url":null,"slug":"instruction-driven-fusion-of-infrared-visible","title":"Instruction-Driven Fusion of Infrared-Visible Images: Tailoring for Diverse Downstream Tasks","date":"2024-11-14","arxiv_id":"2411.09387","repositories_listed":0,"syntology":null},{"url":null,"slug":"leap-d-a-novel-prompt-based-approach-for","title":"LEAP:D - A Novel Prompt-based Approach for Domain-Generalized Aerial Object Detection","date":"2024-11-14","arxiv_id":"2411.09180","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-tailed-object-detection-pre-training","title":"Long-Tailed Object Detection Pre-training: Dynamic Rebalancing Contrastive Learning with Dual Reconstruction","date":"2024-11-14","arxiv_id":"2411.09453","repositories_listed":0,"syntology":null},{"url":null,"slug":"methodology-for-a-statistical-analysis-of","title":"Methodology for a Statistical Analysis of Influencing Factors on 3D Object Detection Performance","date":"2024-11-13","arxiv_id":"2411.08482","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-object-detection-using-depth-and","title":"Multimodal Object Detection using Depth and Image Data for Manufacturing Parts","date":"2024-11-13","arxiv_id":"2411.09062","repositories_listed":0,"syntology":null},{"url":null,"slug":"uiformer-a-unified-transformer-based","title":"UIFormer: A Unified Transformer-based Framework for Incremental Few-Shot Object Detection and Instance Segmentation","date":"2024-11-13","arxiv_id":"2411.08569","repositories_listed":0,"syntology":null},{"url":null,"slug":"depthwise-separable-convolutions-with-deep","title":"Depthwise Separable Convolutions with Deep Residual Convolutions","date":"2024-11-12","arxiv_id":"2411.07544","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-3d-perception-on-multi-sweep-point","title":"Efficient 3D Perception on Multi-Sweep Point Cloud with Gumbel Spatial Pruning","date":"2024-11-12","arxiv_id":"2411.07742","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-scale-frequency-enhancement-network-for","title":"Multi-scale Frequency Enhancement Network for Blind Image Deblurring","date":"2024-11-11","arxiv_id":"2411.06893","repositories_listed":0,"syntology":null},{"url":null,"slug":"track-any-peppers-weakly-supervised-sweet","title":"Track Any Peppers: Weakly Supervised Sweet Pepper Tracking Using VLMs","date":"2024-11-11","arxiv_id":"2411.06702","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-compass-a-comprehensive-and-effective","title":"AI-Compass: A Comprehensive and Effective Multi-module Testing Tool for AI Systems","date":"2024-11-09","arxiv_id":"2411.06146","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-collision-risk-estimation-via","title":"FuzzRisk: Online Collision Risk Estimation for Autonomous Vehicles based on Depth-Aware Object Detection via Fuzzy Inference","date":"2024-11-09","arxiv_id":"2411.08060","repositories_listed":0,"syntology":null},{"url":null,"slug":"pattern-integration-and-enhancement-vision","title":"Pattern Integration and Enhancement Vision Transformer for Self-Supervised Learning in Remote Sensing","date":"2024-11-09","arxiv_id":"2411.06091","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-object-detection-modality-into","title":"Integrating Object Detection Modality into Visual Language Model for Enhanced Autonomous Driving Agent","date":"2024-11-08","arxiv_id":"2411.05898","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-set-object-detection-towards-unified","title":"Open-set object detection: towards unified problem formulation and benchmarking","date":"2024-11-08","arxiv_id":"2411.05564","repositories_listed":0,"syntology":null},{"url":null,"slug":"simplebev-improved-lidar-camera-fusion","title":"SimpleBEV: Improved LiDAR-Camera Fusion Architecture for 3D Object Detection","date":"2024-11-08","arxiv_id":"2411.05292","repositories_listed":0,"syntology":null},{"url":null,"slug":"zopp-a-framework-of-zero-shot-offboard","title":"ZOPP: A Framework of Zero-shot Offboard Panoptic Perception for Autonomous Driving","date":"2024-11-08","arxiv_id":"2411.05311","repositories_listed":0,"syntology":null},{"url":null,"slug":"l0-regularized-sparse-coding-based","title":"l0-Regularized Sparse Coding-based Interpretable Network for Multi-Modal Image Fusion","date":"2024-11-07","arxiv_id":"2411.04519","repositories_listed":0,"syntology":null},{"url":null,"slug":"pose2trajectory-using-transformers-on-body","title":"Pose2Trajectory: Using Transformers on Body Pose to Predict Tennis Player's Trajectory","date":"2024-11-07","arxiv_id":"2411.04501","repositories_listed":0,"syntology":null},{"url":null,"slug":"uevavd-a-dataset-for-developing-uav-s-eye","title":"UEVAVD: A Dataset for Developing UAV's Eye View Active Object Detection","date":"2024-11-07","arxiv_id":"2411.04348","repositories_listed":0,"syntology":null},{"url":null,"slug":"estimation-of-psychosocial-work-environment","title":"Estimation of Psychosocial Work Environment Exposures Through Video Object Detection. Proof of Concept Using CCTV Footage","date":"2024-11-06","arxiv_id":"2411.03724","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-application-agnostic-automatic-target","title":"An Application-Agnostic Automatic Target Recognition System Using Vision Language Models","date":"2024-11-05","arxiv_id":"2411.03491","repositories_listed":0,"syntology":null},{"url":null,"slug":"centerness-based-instance-aware-knowledge","title":"Centerness-based Instance-aware Knowledge Distillation with Task-wise Mutual Lifting for Object Detection on Drone Imagery","date":"2024-11-05","arxiv_id":"2411.02861","repositories_listed":0,"syntology":null},{"url":null,"slug":"erup-yolo-enhancing-object-detection","title":"ERUP-YOLO: Enhancing Object Detection Robustness for Adverse Weather Condition by Unified Image-Adaptive Processing","date":"2024-11-05","arxiv_id":"2411.02799","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-pixels-to-prose-advancing-multi-modal","title":"From Pixels to Prose: Advancing Multi-Modal Language Models for Remote Sensing","date":"2024-11-05","arxiv_id":"2411.05826","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-cross-modality-learning-for","title":"Self-supervised cross-modality learning for uncertainty-aware object detection and recognition in applications which lack pre-labelled training data","date":"2024-11-05","arxiv_id":"2411.03082","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-video-recording-optimization","title":"Intelligent Video Recording Optimization using Activity Detection for Surveillance Systems","date":"2024-11-04","arxiv_id":"2411.02632","repositories_listed":0,"syntology":null},{"url":"/paper/sira-scalable-inter-frame-relation-and-1","slug":"sira-scalable-inter-frame-relation-and-1","title":"SIRA: Scalable Inter-frame Relation and Association for Radar Perception","date":"2024-11-04","arxiv_id":"2411.02220","repositories_listed":0,"syntology":null},{"url":null,"slug":"v-cas-a-realtime-vehicle-anti-collision","title":"V-CAS: A Realtime Vehicle Anti Collision System Using Vision Transformer on Multi-Camera Streams","date":"2024-11-04","arxiv_id":"2411.01963","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-visual-question-answering-method-for-sar","title":"A Visual Question Answering Method for SAR Ship: Breaking the Requirement for Multimodal Dataset Construction and Model Fine-Tuning","date":"2024-11-03","arxiv_id":"2411.01445","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-deep-learning-infrastructures-for","title":"Efficient Deep Learning Infrastructures for Embedded Computing Systems: A Comprehensive Survey and Future Envision","date":"2024-11-03","arxiv_id":"2411.01431","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-for-all-multi-domain-joint-training-for","title":"One for All: Multi-Domain Joint Training for Point Cloud Based 3D Object Detection","date":"2024-11-03","arxiv_id":"2411.01584","repositories_listed":0,"syntology":null}],"record_sha256":"383185ab6f69c05a4dc51a1dc6f4c8daaa704986122aad58313ff9ca649c7a80","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}