{"url":"/task/object-counting","name":"Object Counting","slug":"object-counting","description_markdown":"The goal of **Object Counting** task is to count the number of object instances in a single image or video sequence. It has many real-world applications such as traffic flow monitoring, crowdedness estimation, and product counting.\n\n\n<span class=\"description-source\">Source: [Learning to Count Objects with Few Exemplar Annotations ](https://arxiv.org/abs/1905.07898)</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":158,"papers_with_code":81,"benchmarks":10,"benchmark_tables_in_archive":10,"benchmark_tables_shown":10,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":28,"subtasks":4,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/object-counting-on-fsc147","slug":"object-counting-on-fsc147","dataset":"FSC147","dataset_url":"/dataset/fsc147","rows_in_archive":19,"metrics":["MAE(test)","MAE(val)","RMSE(test)","RMSE(val)"],"first_row_in_archive_order":{"model":"CountGD","paper_title":"CountGD: Multi-Modal Open-World Counting","paper_url":"/paper/countgd-multi-modal-open-world-counting","paper_date":"2024-07-05","arxiv_id":"2407.04619","code_links":[{"title":"niki-amini-naieni/CountGD","url":"https://github.com/niki-amini-naieni/CountGD"},{"title":"niki-amini-naieni/countx","url":"https://github.com/niki-amini-naieni/countx"}],"syntology":{"n":9,"n_ran":4,"n_unverified":5,"n_pointer_only":0}}},{"leaderboard":"/sota/object-counting-on-carpk","slug":"object-counting-on-carpk","dataset":"CARPK","dataset_url":"/dataset/carpk","rows_in_archive":15,"metrics":["MAE","RMSE"],"first_row_in_archive_order":{"model":"HLCNN","paper_title":"An Accurate Car Counting in Aerial Images Based on Convolutional Neural Networks","paper_url":"/paper/an-accurate-car-counting-in-aerial-images","paper_date":"2021-07-13","arxiv_id":null,"code_links":[{"title":"ekilic/Heatmap-Learner-CNN-for-Object-Counting","url":"https://github.com/ekilic/Heatmap-Learner-CNN-for-Object-Counting"}],"syntology":null}},{"leaderboard":"/sota/object-counting-on-pascal-voc-2007-count-test","slug":"object-counting-on-pascal-voc-2007-count-test","dataset":"Pascal VOC 2007 count-test","dataset_url":"/dataset/pascal-voc-2007","rows_in_archive":8,"metrics":["m-reIRMSE-nz","m-relRMSE","mRMSE","mRMSE-nz"],"first_row_in_archive_order":{"model":"Supervised Density Map","paper_title":"Object Counting and Instance Segmentation with Image-level Supervision","paper_url":"/paper/object-counting-and-instance-segmentation","paper_date":"2019-03-06","arxiv_id":"1903.02494","code_links":[{"title":"GuoleiSun/CountSeg","url":"https://github.com/GuoleiSun/CountSeg"},{"title":"alzayats/CountSeg-1","url":"https://github.com/alzayats/CountSeg-1"}],"syntology":null}},{"leaderboard":"/sota/object-counting-on-coco-count-test","slug":"object-counting-on-coco-count-test","dataset":"COCO count-test","dataset_url":"/dataset/coco","rows_in_archive":7,"metrics":["m-reIRMSE","m-reIRMSE-nz","mRMSE","mRMSE-nz"],"first_row_in_archive_order":{"model":"ens","paper_title":"Counting Everyday Objects in Everyday Scenes","paper_url":"/paper/counting-everyday-objects-in-everyday-scenes","paper_date":"2016-04-12","arxiv_id":"1604.03505","code_links":[{"title":"prithv1/cvpr2017_counting","url":"https://github.com/prithv1/cvpr2017_counting"}],"syntology":null}},{"leaderboard":"/sota/object-counting-on-tallyqa-complex","slug":"object-counting-on-tallyqa-complex","dataset":"TallyQA-Complex","dataset_url":"/dataset/tallyqa","rows_in_archive":6,"metrics":["Accuracy","RMSE"],"first_row_in_archive_order":{"model":"SMoLA-PaLI-X Specialist","paper_title":"Omni-SMoLA: Boosting Generalist Multimodal Models with Soft Mixture of Low-rank Experts","paper_url":"/paper/omni-smola-boosting-generalist-multimodal","paper_date":"2023-12-01","arxiv_id":"2312.00968","code_links":[],"syntology":null}},{"leaderboard":"/sota/object-counting-on-tallyqa-simple","slug":"object-counting-on-tallyqa-simple","dataset":"TallyQA-Simple","dataset_url":"/dataset/tallyqa","rows_in_archive":6,"metrics":["Accuracy","RMSE"],"first_row_in_archive_order":{"model":"SMoLA-PaLI-X Specialist","paper_title":"Omni-SMoLA: Boosting Generalist Multimodal Models with Soft Mixture of Low-rank Experts","paper_url":"/paper/omni-smola-boosting-generalist-multimodal","paper_date":"2023-12-01","arxiv_id":"2312.00968","code_links":[],"syntology":null}},{"leaderboard":"/sota/object-counting-on-howmany-qa","slug":"object-counting-on-howmany-qa","dataset":"HowMany-QA","dataset_url":"/dataset/howmany-qa","rows_in_archive":3,"metrics":["Accuracy","RMSE"],"first_row_in_archive_order":{"model":"MoVie-ResNeXt","paper_title":"MoVie: Revisiting Modulated Convolutions for Visual Counting and Beyond","paper_url":"/paper/revisiting-modulated-convolutions-for-visual","paper_date":"2020-04-24","arxiv_id":"2004.11883","code_links":[{"title":"facebookresearch/mmf","url":"https://github.com/facebookresearch/mmf/tree/master/projects/movie_mcan"}],"syntology":null}},{"leaderboard":"/sota/object-counting-on-pascal-voc","slug":"object-counting-on-pascal-voc","dataset":"PASCAL VOC","dataset_url":"/dataset/pascal-voc","rows_in_archive":3,"metrics":["mRMSE"],"first_row_in_archive_order":{"model":"TFOC","paper_title":"Training-free Object Counting with Prompts","paper_url":"/paper/training-free-object-counting-with-prompts","paper_date":"2023-06-30","arxiv_id":"2307.00038","code_links":[{"title":"shizenglin/training-free-object-counter","url":"https://github.com/shizenglin/training-free-object-counter"}],"syntology":null}},{"leaderboard":"/sota/object-counting-on-omnicount-191","slug":"object-counting-on-omnicount-191","dataset":"Omnicount-191","dataset_url":"/dataset/omnicount-191","rows_in_archive":1,"metrics":["mRMSE"],"first_row_in_archive_order":{"model":"Omnicount","paper_title":"OmniCount: Multi-label Object Counting with Semantic-Geometric Priors","paper_url":"/paper/omnicount-multi-label-object-counting-with","paper_date":"2024-03-08","arxiv_id":"2403.05435","code_links":[],"syntology":null}},{"leaderboard":"/sota/object-counting-on-trancos","slug":"object-counting-on-trancos","dataset":"TRANCOS","dataset_url":"/dataset/trancos","rows_in_archive":1,"metrics":["MAE","MSE"],"first_row_in_archive_order":{"model":"GauNet (ResNet-50)","paper_title":"Rethinking Spatial Invariance of Convolutional Networks for Object Counting","paper_url":"/paper/rethinking-spatial-invariance-of-1","paper_date":"2022-06-10","arxiv_id":"2206.05253","code_links":[{"title":"zhiqic/rethinking-counting","url":"https://github.com/zhiqic/rethinking-counting"}],"syntology":null}}],"datasets":[{"url":"/dataset/coco","name":"COCO (Common Objects in Context)","full_name":"Common Objects in Context","num_papers_in_archive":11922},{"url":"/dataset/pascal-voc","name":"PASCAL VOC","full_name":"PASCAL Visual Object Classes Challenge","num_papers_in_archive":198},{"url":"/dataset/pascal-voc-2007","name":"PASCAL VOC 2007","full_name":"PASCAL VOC 2007","num_papers_in_archive":126},{"url":"/dataset/carpk","name":"CARPK","full_name":"car parking lot dataset","num_papers_in_archive":71},{"url":"/dataset/fsod","name":"FSOD","full_name":"Few-Shot Object Detection Dataset","num_papers_in_archive":68},{"url":"/dataset/fsc147","name":"FSC147","full_name":"","num_papers_in_archive":58},{"url":"/dataset/tallyqa","name":"TallyQA","full_name":"","num_papers_in_archive":33},{"url":"/dataset/cowc","name":"COWC","full_name":"Cars Overhead With Context","num_papers_in_archive":22},{"url":"/dataset/cost","name":"COST","full_name":"COCO Segmentation Text","num_papers_in_archive":11},{"url":"/dataset/agar","name":"AGAR","full_name":"Annotated Germs for Automated Recognition","num_papers_in_archive":7},{"url":"/dataset/trancos","name":"TRANCOS","full_name":"TRaffic ANd COngestionS","num_papers_in_archive":7},{"url":"/dataset/25ktrees","name":"25kTrees","full_name":"Individual Tree Crown Annotations","num_papers_in_archive":5},{"url":"/dataset/howmany-qa","name":"HowMany-QA","full_name":"","num_papers_in_archive":5},{"url":"/dataset/rf100","name":"RF100","full_name":"Roboflow 100","num_papers_in_archive":5},{"url":"/dataset/fluocells","name":"fluocells","full_name":"Fluorescent Neuronal Cells","num_papers_in_archive":3},{"url":"/dataset/iwildcam-2021","name":"iWildCam 2021","full_name":"","num_papers_in_archive":3},{"url":"/dataset/blue-cells-enumeration-dataset","name":"Blue Cells enumeration dataset","full_name":"Learning to Count Objects in Images blue cells dataset","num_papers_in_archive":2},{"url":"/dataset/leukemiaattri","name":"LeukemiaAttri","full_name":"","num_papers_in_archive":2},{"url":"/dataset/rsoc","name":"RSOC","full_name":"Remote Sensing Object Counting","num_papers_in_archive":2},{"url":"/dataset/smartcity","name":"SmartCity","full_name":"","num_papers_in_archive":2},{"url":"/dataset/acct-data-repository","name":"ACCT Data Repository","full_name":"ACCT is a fast and accessible automatic cell counting tool using machine learning for 2D image segmentation","num_papers_in_archive":1},{"url":"/dataset/artificial-fluorescent-bacteria-dataset","name":"Artificial fluorescent bacteria dataset","full_name":"Artificial fluorescent bacteria dataset","num_papers_in_archive":1},{"url":"/dataset/bbbc041","name":"BBBC041","full_name":"P. vivax (malaria) infected human blood smears","num_papers_in_archive":1},{"url":"/dataset/fish-counting","name":"Fish Counting","full_name":"","num_papers_in_archive":1},{"url":"/dataset/locount","name":"Locount","full_name":"","num_papers_in_archive":1},{"url":"/dataset/omnicount-191","name":"Omnicount-191","full_name":"","num_papers_in_archive":1},{"url":"/dataset/vgg-cell","name":"VGG Cell","full_name":null,"num_papers_in_archive":1},{"url":"/dataset/penguin-dataset","name":"Penguin dataset","full_name":"","num_papers_in_archive":0}],"subtasks":[{"url":"/task/exemplar-free-counting","name":"Exemplar-Free Counting"},{"url":"/task/few-shot-object-counting-and-detection","name":"Few-shot Object Counting and Detection"},{"url":"/task/open-vocabulary-object-counting","name":"Open-vocabulary object counting"},{"url":"/task/training-free-object-counting","name":"Training-free Object Counting"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":81,"tagged_in_all":158,"items":[{"url":"/paper/yolo9000-better-faster-stronger","title":"YOLO9000: Better, Faster, Stronger","date":"2016-12-25","arxiv_id":"1612.08242","repositories_listed":231,"syntology":{"n":60,"n_ran":16,"n_unverified":44,"n_pointer_only":22}},{"url":"/paper/you-only-look-once-unified-real-time-object","title":"You Only Look Once: Unified, Real-Time Object Detection","date":"2015-06-08","arxiv_id":"1506.02640","repositories_listed":144,"syntology":{"n":148,"n_ran":80,"n_unverified":68,"n_pointer_only":98}},{"url":"/paper/self-supervised-learning-from-images-with-a","title":"Self-Supervised Learning from Images with a Joint-Embedding Predictive Architecture","date":"2023-01-19","arxiv_id":"2301.08243","repositories_listed":7,"syntology":{"n":14,"n_ran":5,"n_unverified":9,"n_pointer_only":13}},{"url":"/paper/synbols-probing-learning-algorithms-with","title":"Synbols: Probing Learning Algorithms with Synthetic Datasets","date":"2020-09-14","arxiv_id":"2009.06415","repositories_listed":4,"syntology":null},{"url":"/paper/visionreasoner-unified-visual-perception-and","title":"VisionReasoner: Unified Visual Perception and Reasoning via Reinforcement Learning","date":"2025-05-17","arxiv_id":"2505.12081","repositories_listed":3,"syntology":{"n":15,"n_ran":1,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/real-time-pear-fruit-detection-and-counting","title":"Real Time Pear Fruit Detection and Counting Using YOLOv4 Models and Deep SORT","date":"2021-07-14","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/cnn-based-density-estimation-and-crowd","title":"CNN-based Density Estimation and Crowd Counting: A Survey","date":"2020-03-28","arxiv_id":"2003.12783","repositories_listed":3,"syntology":null},{"url":"/paper/from-open-set-to-closed-set-supervised","title":"From Open Set to Closed Set: Supervised Spatial Divide-and-Conquer for Object Counting","date":"2020-01-07","arxiv_id":"2001.01886","repositories_listed":3,"syntology":null},{"url":"/paper/where-are-the-blobs-counting-by-localization","title":"Where are the Blobs: Counting by Localization with Point Supervision","date":"2018-07-25","arxiv_id":"1807.09856","repositories_listed":3,"syntology":{"n":12,"n_ran":0,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/vision-transformers-for-weakly-supervised","title":"Vision Transformers for Weakly-Supervised Microorganism Enumeration","date":"2024-12-03","arxiv_id":"2412.02250","repositories_listed":2,"syntology":null},{"url":"/paper/countgd-multi-modal-open-world-counting","title":"CountGD: Multi-Modal Open-World Counting","date":"2024-07-05","arxiv_id":"2407.04619","repositories_listed":2,"syntology":{"n":9,"n_ran":4,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/learning-to-count-anything-reference-less","title":"Learning to Count Anything: Reference-less Class-agnostic Counting with Weak Supervision","date":"2022-05-20","arxiv_id":"2205.10203","repositories_listed":2,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/dall-eval-probing-the-reasoning-skills-and","title":"DALL-Eval: Probing the Reasoning Skills and Social Biases of Text-to-Image Generation Models","date":"2022-02-08","arxiv_id":"2202.04053","repositories_listed":2,"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/unsupervised-domain-adaptation-for-plant","title":"Unsupervised Domain Adaptation For Plant Organ Counting","date":"2020-09-02","arxiv_id":"2009.01081","repositories_listed":2,"syntology":null},{"url":"/paper/encoder-decoder-based-convolutional-neural","title":"Encoder-Decoder Based Convolutional Neural Networks with Multi-Scale-Aware Modules for Crowd Counting","date":"2020-03-12","arxiv_id":"2003.05586","repositories_listed":2,"syntology":null},{"url":"/paper/drone-based-rgbt-vehicle-detection-and","title":"Drone-based RGB-Infrared Cross-Modality Vehicle Detection via Uncertainty-Aware Learning","date":"2020-03-05","arxiv_id":"2003.02437","repositories_listed":2,"syntology":null},{"url":"/paper/object-counting-and-instance-segmentation","title":"Object Counting and Instance Segmentation with Image-level Supervision","date":"2019-03-06","arxiv_id":"1903.02494","repositories_listed":2,"syntology":null},{"url":"/paper/improving-object-counting-with-heatmap","title":"Improving Object Counting with Heatmap Regulation","date":"2018-03-14","arxiv_id":"1803.05494","repositories_listed":2,"syntology":null},{"url":"/paper/towards-perspective-free-object-counting-with","title":"Towards perspective-free object counting with deep learning","date":"2016-01-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/car-object-counting-and-position-estimation","title":"Car Object Counting and Position Estimation via Extension of the CLIP-EBC Framework","date":"2025-07-11","arxiv_id":"2507.08240","repositories_listed":1,"syntology":null},{"url":"/paper/improving-contrastive-learning-for-referring","title":"Improving Contrastive Learning for Referring Expression Counting","date":"2025-05-28","arxiv_id":"2505.22850","repositories_listed":1,"syntology":null},{"url":"/paper/instructsam-a-training-free-framework-for","title":"InstructSAM: A Training-Free Framework for Instruction-Oriented Remote Sensing Object Recognition","date":"2025-05-21","arxiv_id":"2505.15818","repositories_listed":1,"syntology":null},{"url":"/paper/save-self-attention-on-visual-embedding-for","title":"SAVE: Self-Attention on Visual Embedding for Zero-Shot Generic Object Counting","date":"2025-02-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/t2icount-enhancing-cross-modal-understanding","title":"T2ICount: Enhancing Cross-modal Understanding for Zero-Shot Counting","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/geobench-vlm-benchmarking-vision-language","title":"GEOBench-VLM: Benchmarking Vision-Language Models for Geospatial Tasks","date":"2024-11-28","arxiv_id":"2411.19325","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/a-novel-unified-architecture-for-low-shot","title":"A Novel Unified Architecture for Low-Shot Counting by Detection and Segmentation","date":"2024-09-27","arxiv_id":"2409.18686","repositories_listed":1,"syntology":null},{"url":"/paper/mind-the-prompt-a-novel-benchmark-for-prompt","title":"Mind the Prompt: A Novel Benchmark for Prompt-based Class-Agnostic Counting","date":"2024-09-24","arxiv_id":"2409.15953","repositories_listed":1,"syntology":null},{"url":"/paper/dense-center-direction-regression-for-object","title":"Dense Center-Direction Regression for Object Counting and Localization with Point Supervision","date":"2024-08-26","arxiv_id":"2408.14457","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-object-counting-with-good-exemplars","title":"Zero-shot Object Counting with Good Exemplars","date":"2024-07-06","arxiv_id":"2407.04948","repositories_listed":1,"syntology":null},{"url":"/paper/rs-agent-automating-remote-sensing-tasks","title":"RS-Agent: Automating Remote Sensing Tasks through Intelligent Agent","date":"2024-06-11","arxiv_id":"2406.07089","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":0}}],"syntology_records":10,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}