{"url":"/sota/object-counting-on-carpk","task":{"name":"Object Counting","url":"/task/object-counting","note":null},"dataset":{"name":"CARPK","url":"/dataset/carpk"},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"The goal of **Object Counting** task is to count the number of object instances in a single image or video sequence. It has many real-world applications such as traffic flow monitoring, crowdedness estimation, and product counting.\n\n\n<span class=\"description-source\">Source: [Learning to Count Objects with Few Exemplar Annotations ](https://arxiv.org/abs/1905.07898)</span>","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["MAE","RMSE"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"MAE":"lower","RMSE":"lower"}},"counts":{"rows":15,"rows_with_code":12,"rows_with_paper_page":15,"rows_dated":15,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"HLCNN","metrics":{"MAE":"2.12","RMSE":"3.02"},"uses_additional_data":false,"paper_date":"2021-07-13","paper":"/paper/an-accurate-car-counting-in-aerial-images","paper_url":"https://link.springer.com/article/10.1007%2Fs12652-021-03377-5","paper_title":"An Accurate Car Counting in Aerial Images Based on Convolutional Neural Networks","code":"https://github.com/ekilic/Heatmap-Learner-CNN-for-Object-Counting","n_code_links":1,"syntology":null},{"rank_in_archive_order":2,"model":"CLIP-LOCAR","metrics":{"MAE":"4.01","RMSE":"6.02"},"uses_additional_data":false,"paper_date":"2025-07-11","paper":"/paper/car-object-counting-and-position-estimation","paper_url":"https://arxiv.org/abs/2507.08240v1","paper_title":"Car Object Counting and Position Estimation via Extension of the CLIP-EBC Framework","code":"https://github.com/jungseoik/CLIP-LOCAR","n_code_links":1,"syntology":null},{"rank_in_archive_order":3,"model":"SAFECount","metrics":{"MAE":"5.33","RMSE":"7.04"},"uses_additional_data":false,"paper_date":"2022-01-22","paper":"/paper/iterative-correlation-based-feature","paper_url":"https://arxiv.org/abs/2201.08959v5","paper_title":"Few-shot Object Counting with Similarity-Aware Feature Enhancement","code":"https://github.com/zhiyuanyou/SAFECount","n_code_links":1,"syntology":null},{"rank_in_archive_order":4,"model":"CounTR","metrics":{"MAE":"5.75","RMSE":"7.45"},"uses_additional_data":false,"paper_date":"2022-08-29","paper":"/paper/countr-transformer-based-generalised-visual","paper_url":"https://arxiv.org/abs/2208.13721v3","paper_title":"CounTR: Transformer-based Generalised Visual Counting","code":"https://github.com/Verg-Avesta/CounTR","n_code_links":1,"syntology":{"n_ran":1,"n_unverified":3,"n_samples":4,"n_pointer_only_licence":0}},{"rank_in_archive_order":5,"model":"BMNet+","metrics":{"MAE":"5.76","RMSE":"7.83"},"uses_additional_data":false,"paper_date":"2022-03-16","paper":"/paper/represent-compare-and-learn-a-similarity","paper_url":"https://arxiv.org/abs/2203.08354v1","paper_title":"Represent, Compare, and Learn: A Similarity-Aware Framework for Class-Agnostic Counting","code":"https://github.com/flyinglynx/Bilinear-Matching-Network","n_code_links":1,"syntology":{"n_ran":0,"n_unverified":1,"n_samples":1,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"VLCounter","metrics":{"MAE":"6.46","RMSE":"8.68"},"uses_additional_data":false,"paper_date":"2023-12-27","paper":"/paper/vlcounter-text-aware-visual-representation","paper_url":"https://arxiv.org/abs/2312.16580v2","paper_title":"VLCounter: Text-aware Visual Representation for Zero-Shot Object Counting","code":"https://github.com/seunggu0305/vlcounter","n_code_links":1,"syntology":null},{"rank_in_archive_order":7,"model":"Soft-IoU + EM-Merger unit","metrics":{"MAE":"6.77","RMSE":"8.52"},"uses_additional_data":false,"paper_date":"2019-04-01","paper":"/paper/precise-detection-in-densely-packed-scenes","paper_url":"http://arxiv.org/abs/1904.00853v3","paper_title":"Precise Detection in Densely Packed Scenes","code":"https://github.com/eg4000/SKU110K_CVPR19","n_code_links":5,"syntology":null},{"rank_in_archive_order":8,"model":"CounTX (uses arbitrary text input to specify object to count, used \"the cars\" for CARPK)","metrics":{"MAE":"8.13","RMSE":"10.87"},"uses_additional_data":false,"paper_date":"2023-06-02","paper":"/paper/open-world-text-specified-object-counting","paper_url":"https://arxiv.org/abs/2306.01851v2","paper_title":"Open-world Text-specified Object Counting","code":"https://github.com/niki-amini-naieni/countx","n_code_links":1,"syntology":null},{"rank_in_archive_order":9,"model":"RetinaNet (2018)","metrics":{"MAE":"16.62","RMSE":"22.30"},"uses_additional_data":false,"paper_date":"2017-07-19","paper":"/paper/drone-based-object-counting-by-spatially","paper_url":"http://arxiv.org/abs/1707.05972v3","paper_title":"Drone-based Object Counting by Spatially Regularized Regional Proposal Network","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":10,"model":"One-Look Regression (2016)","metrics":{"MAE":"21.88","RMSE":"36.73"},"uses_additional_data":false,"paper_date":"2016-09-14","paper":"/paper/a-large-contextual-dataset-for-classification","paper_url":"http://arxiv.org/abs/1609.04453v1","paper_title":"A Large Contextual Dataset for Classification, Detection and Counting of Cars with Deep Learning","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":11,"model":"LPN Counting (2017)","metrics":{"MAE":"22.76","RMSE":"34.46"},"uses_additional_data":false,"paper_date":"2017-07-19","paper":"/paper/drone-based-object-counting-by-spatially","paper_url":"http://arxiv.org/abs/1707.05972v3","paper_title":"Drone-based Object Counting by Spatially Regularized Regional Proposal Network","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":12,"model":"RetinaNet (2018)","metrics":{"MAE":"24.58"},"uses_additional_data":false,"paper_date":"2017-08-07","paper":"/paper/focal-loss-for-dense-object-detection","paper_url":"http://arxiv.org/abs/1708.02002v2","paper_title":"Focal Loss for Dense Object Detection","code":"https://github.com/tensorflow/models","n_code_links":234,"syntology":{"n_ran":11,"n_unverified":0,"n_samples":11,"n_pointer_only_licence":6}},{"rank_in_archive_order":13,"model":"Faster R-CNN (2015)","metrics":{"MAE":"39.88","RMSE":"47.67"},"uses_additional_data":false,"paper_date":"2015-06-04","paper":"/paper/faster-r-cnn-towards-real-time-object","paper_url":"http://arxiv.org/abs/1506.01497v3","paper_title":"Faster R-CNN: Towards Real-Time Object Detection with Region Proposal Networks","code":"https://github.com/facebookresearch/detectron2","n_code_links":196,"syntology":{"n_ran":59,"n_unverified":65,"n_samples":124,"n_pointer_only_licence":42}},{"rank_in_archive_order":14,"model":"YOLO9000opt (2017)","metrics":{"MAE":"130.40","RMSE":"172.46"},"uses_additional_data":false,"paper_date":"2016-12-25","paper":"/paper/yolo9000-better-faster-stronger","paper_url":"http://arxiv.org/abs/1612.08242v1","paper_title":"YOLO9000: Better, Faster, Stronger","code":"https://github.com/AlexeyAB/darknet","n_code_links":231,"syntology":{"n_ran":16,"n_unverified":44,"n_samples":60,"n_pointer_only_licence":22}},{"rank_in_archive_order":15,"model":"YOLO (2016)","metrics":{"MAE":"156.00","RMSE":"57.55"},"uses_additional_data":false,"paper_date":"2015-06-08","paper":"/paper/you-only-look-once-unified-real-time-object","paper_url":"http://arxiv.org/abs/1506.02640v5","paper_title":"You Only Look Once: Unified, Real-Time Object Detection","code":"https://github.com/AlexeyAB/darknet","n_code_links":144,"syntology":{"n_ran":80,"n_unverified":68,"n_samples":148,"n_pointer_only_licence":98}}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":6,"rows_with_any_sample_ran":5,"distinct_papers_with_graph_line":6,"distinct_papers_with_any_sample_ran":5,"samples_over_distinct_papers":{"n_ran":167,"n_unverified":181,"n_samples":348,"n_pointer_only_licence":168,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":167,"n_unverified":181,"n_samples":348,"n_pointer_only_licence":168,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}