{"url":"/task/multispectral-object-detection","name":"Multispectral Object Detection","slug":"multispectral-object-detection","description_markdown":"Only using RGB cameras for automatic outdoor scene analysis is challenging when, for example, facing insufficient illumination or adverse weather. To improve the recognition reliability, multispectral systems add additional cameras (e.g. infra-red) and perform object detection from multispectral data. Although multispectral scene analysis with deep learning has be shown to have a great potential, there are still many open research questions and it has not been widely deployed in industrial contexts.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":39,"papers_with_code":26,"benchmarks":4,"benchmark_tables_in_archive":4,"benchmark_tables_shown":4,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/multispectral-object-detection-on-flir-1","slug":"multispectral-object-detection-on-flir-1","dataset":"FLIR","dataset_url":null,"rows_in_archive":18,"metrics":["mAP50","mAP"],"first_row_in_archive_order":{"model":"MMPedestron","paper_title":"When Pedestrian Detection Meets Multi-Modal Learning: Generalist Model and Benchmark Dataset","paper_url":"/paper/when-pedestrian-detection-meets-multi-modal","paper_date":"2024-07-14","arxiv_id":"2407.10125","code_links":[{"title":"BubblyYi/MMPedestron","url":"https://github.com/BubblyYi/MMPedestron"}],"syntology":null}},{"leaderboard":"/sota/multispectral-object-detection-on-kaist","slug":"multispectral-object-detection-on-kaist","dataset":"KAIST Multispectral Pedestrian Detection Benchmark","dataset_url":"/dataset/kaist-multispectral-pedestrian-detection","rows_in_archive":17,"metrics":["All Miss Rate","Reasonable Miss Rate"],"first_row_in_archive_order":{"model":"RSDet","paper_title":"Removal then Selection: A Coarse-to-Fine Fusion Perspective for RGB-Infrared Object Detection","paper_url":"/paper/removal-and-selection-improving-rgb-infrared","paper_date":"2024-01-19","arxiv_id":"2401.10731","code_links":[{"title":"Zhao-Tian-yi/RSDet","url":"https://github.com/Zhao-Tian-yi/RSDet"}],"syntology":null}},{"leaderboard":"/sota/multispectral-object-detection-on-nii-cu-mapd","slug":"multispectral-object-detection-on-nii-cu-mapd","dataset":"NII-CU MAPD","dataset_url":"/dataset/nii-cu-mapd","rows_in_archive":2,"metrics":["mAP@0.5:0.95","AP@0.5","AP@0.75"],"first_row_in_archive_order":{"model":"YOLOv3-4‐channel","paper_title":"Deep learning with RGB and thermal images onboard a drone for monitoring operations","paper_url":"/paper/deep-learning-with-rgb-and-thermal-images","paper_date":"2022-05-31","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/multispectral-object-detection-on-llvip","slug":"multispectral-object-detection-on-llvip","dataset":"LLVIP","dataset_url":"/dataset/llvip","rows_in_archive":1,"metrics":["mAP50"],"first_row_in_archive_order":{"model":"CFT","paper_title":"Cross-Modality Fusion Transformer for Multispectral Object Detection","paper_url":"/paper/cross-modality-fusion-transformer-for","paper_date":"2021-10-30","arxiv_id":"2111.00273","code_links":[{"title":"docf/multispectral-object-detection","url":"https://github.com/docf/multispectral-object-detection"}],"syntology":null}}],"datasets":[{"url":"/dataset/llvip","name":"LLVIP","full_name":"A Visible-infrared Paired Dataset for Low-light Vision","num_papers_in_archive":116},{"url":"/dataset/kaist-multispectral-pedestrian-detection","name":"KAIST Multispectral Pedestrian Detection Benchmark","full_name":"","num_papers_in_archive":23},{"url":"/dataset/nii-cu-mapd","name":"NII-CU MAPD","full_name":"NII-CU Multispectral Aerial Person Detection Dataset","num_papers_in_archive":1},{"url":"/dataset/ms-evs-dataset","name":"MS-EVS Dataset","full_name":"Multispectral Event-based Face detection dataset","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":26,"of":26,"tagged_in_all":39,"items":[{"url":"/paper/fully-convolutional-networks-for-semantic-1","title":"Fully Convolutional Networks for Semantic Segmentation","date":"2014-11-14","arxiv_id":"1411.4038","repositories_listed":51,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/mathbf-c-2-former-calibrated-and","title":"$\\mathbf{C}^2$Former: Calibrated and Complementary Transformer for RGB-Infrared Object Detection","date":"2023-06-28","arxiv_id":"2306.16175","repositories_listed":2,"syntology":null},{"url":"/paper/improving-multispectral-pedestrian-detection","title":"Improving Multispectral Pedestrian Detection by Addressing Modality Imbalance Problems","date":"2020-08-07","arxiv_id":"2008.03043","repositories_listed":2,"syntology":null},{"url":"/paper/multispectral-deep-neural-networks-for","title":"Multispectral Deep Neural Networks for Pedestrian Detection","date":"2016-11-08","arxiv_id":"1611.02644","repositories_listed":2,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/yolov11-rgbt-towards-a-comprehensive-single","title":"YOLOv11-RGBT: Towards a Comprehensive Single-Stage Multispectral Object Detection Framework","date":"2025-06-17","arxiv_id":"2506.14696","repositories_listed":1,"syntology":null},{"url":"/paper/multispectral-detection-transformer-with","title":"Multispectral Detection Transformer with Infrared-Centric Sensor Fusion","date":"2025-05-21","arxiv_id":"2505.15137","repositories_listed":1,"syntology":null},{"url":"/paper/when-pedestrian-detection-meets-multi-modal","title":"When Pedestrian Detection Meets Multi-Modal Learning: Generalist Model and Benchmark Dataset","date":"2024-07-14","arxiv_id":"2407.10125","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-early-fusion-strategies-for","title":"Rethinking Early-Fusion Strategies for Improved Multispectral Object Detection","date":"2024-05-25","arxiv_id":"2405.16038","repositories_listed":1,"syntology":null},{"url":"/paper/mipa-mixed-patch-infrared-visible-modality","title":"MiPa: Mixed Patch Infrared-Visible Modality Agnostic Object Detection","date":"2024-04-29","arxiv_id":"2404.18849","repositories_listed":1,"syntology":null},{"url":"/paper/unirgb-ir-a-unified-framework-for-visible","title":"UniRGB-IR: A Unified Framework for RGB-Infrared Semantic Tasks via Adapter Tuning","date":"2024-04-26","arxiv_id":"2404.17360","repositories_listed":1,"syntology":null},{"url":"/paper/cfmw-cross-modality-fusion-mamba-for","title":"CFMW: Cross-modality Fusion Mamba for Multispectral Object Detection under Adverse Weather Conditions","date":"2024-04-25","arxiv_id":"2404.16302","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_unverified":3,"n_pointer_only":7}},{"url":"/paper/insanet-intra-inter-spectral-attention","title":"INSANet: INtra-INter Spectral Attention Network for Effective Feature Fusion of Multispectral Pedestrian Detection","date":"2024-02-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/removal-and-selection-improving-rgb-infrared","title":"Removal then Selection: A Coarse-to-Fine Fusion Perspective for RGB-Infrared Object Detection","date":"2024-01-19","arxiv_id":"2401.10731","repositories_listed":1,"syntology":null},{"url":"/paper/rgb-x-object-detection-via-scene-specific","title":"RGB-X Object Detection via Scene-Specific Fusion Modules","date":"2023-10-30","arxiv_id":"2310.19372","repositories_listed":1,"syntology":null},{"url":"/paper/icafusion-iterative-cross-attention-guided","title":"ICAFusion: Iterative Cross-Attention Guided Feature Fusion for Multispectral Object Detection","date":"2023-08-15","arxiv_id":"2308.07504","repositories_listed":1,"syntology":null},{"url":"/paper/tfdet-target-aware-fusion-for-rgb-t","title":"TFDet: Target-Aware Fusion for RGB-T Pedestrian Detection","date":"2023-05-26","arxiv_id":"2305.16580","repositories_listed":1,"syntology":null},{"url":"/paper/confidence-aware-fusion-using-dempster-shafer","title":"Confidence-aware Fusion using Dempster-Shafer Theory for Multispectral Pedestrian Detection","date":"2022-03-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cmx-cross-modal-fusion-for-rgb-x-semantic","title":"CMX: Cross-Modal Fusion for RGB-X Semantic Segmentation with Transformers","date":"2022-03-09","arxiv_id":"2203.04838","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/adjust-a-dictionary-based-joint","title":"ADJUST: A Dictionary-Based Joint Reconstruction and Unmixing Method for Spectral Tomography","date":"2021-12-21","arxiv_id":"2112.11406","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modality-fusion-transformer-for","title":"Cross-Modality Fusion Transformer for Multispectral Object Detection","date":"2021-10-30","arxiv_id":"2111.00273","repositories_listed":1,"syntology":null},{"url":"/paper/llvip-a-visible-infrared-paired-dataset-for","title":"LLVIP: A Visible-infrared Paired Dataset for Low-light Vision","date":"2021-08-24","arxiv_id":"2108.10831","repositories_listed":1,"syntology":null},{"url":"/paper/mlpd-multi-label-pedestrian-detector-in","title":"MLPD: Multi-Label Pedestrian Detector in Multispectral Domain","date":"2021-07-26","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/guided-attentive-feature-fusion-for","title":"Guided Attentive Feature Fusion for Multispectral Pedestrian Detection","date":"2021-01-03","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multispectral-fusion-for-object-detection","title":"Multispectral Fusion for Object Detection with Cyclic Fuse-and-Refine Blocks","date":"2020-09-26","arxiv_id":"2009.12664","repositories_listed":1,"syntology":null},{"url":"/paper/cian-cross-image-affinity-net-for-weakly","title":"CIAN: Cross-Image Affinity Net for Weakly Supervised Semantic Segmentation","date":"2018-11-27","arxiv_id":"1811.10842","repositories_listed":1,"syntology":null},{"url":"/paper/multispectral-pedestrian-detection-via","title":"Multispectral Pedestrian Detection via Simultaneous Detection and Segmentation","date":"2018-08-14","arxiv_id":"1808.04818","repositories_listed":1,"syntology":null}],"syntology_records":4,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}