{"url":"/sota/described-object-detection-on-description","task":{"name":"Described Object Detection","url":"/task/described-object-detection","note":null},"dataset":{"name":"Description Detection Dataset","url":"/dataset/description-detection-dataset"},"category":"Computer Vision","categories":["Computer Vision"],"category_note":null,"description":"Described Object Detection (DOD) detects all instances on each image in the dataset, based on a flexible reference. It is a superset of Open-Vocabulary Object Detection (OVD) and Referring Expression Comprehension (REC). It expands category names to flexible language expressions for OVD and overcomes the limitation of REC only grounding the pre-existing object. Works related to DOD are tracked in [awesome-DOD list on github](https://github.com/Charles-Xie/awesome-described-object-detection).","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["Intra-scenario FULL mAP","Intra-scenario PRES mAP","Intra-scenario ABS mAP"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"Intra-scenario FULL mAP":"higher","Intra-scenario PRES mAP":"higher","Intra-scenario ABS mAP":"higher"}},"counts":{"rows":8,"rows_with_code":8,"rows_with_paper_page":8,"rows_dated":8,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"MM-Grounding-DINO","metrics":{"Intra-scenario ABS mAP":"26.0","Intra-scenario FULL mAP":"22.9","Intra-scenario PRES mAP":"21.9"},"uses_additional_data":false,"paper_date":"2024-01-04","paper":"/paper/an-open-and-comprehensive-pipeline-for","paper_url":"https://arxiv.org/abs/2401.02361v2","paper_title":"An Open and Comprehensive Pipeline for Unified Object Grounding and Detection","code":"https://github.com/open-mmlab/mmdetection","n_code_links":2,"syntology":{"n_ran":4,"n_unverified":2,"n_samples":6,"n_pointer_only_licence":0}},{"rank_in_archive_order":2,"model":"FIBER-B","metrics":{"Intra-scenario ABS mAP":"26.0","Intra-scenario FULL mAP":"22.7","Intra-scenario PRES mAP":"21.5"},"uses_additional_data":false,"paper_date":"2022-06-15","paper":"/paper/coarse-to-fine-vision-language-pre-training","paper_url":"https://arxiv.org/abs/2206.07643v2","paper_title":"Coarse-to-Fine Vision-Language Pre-training with Fusion in the Backbone","code":"https://github.com/microsoft/fiber","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":1,"n_samples":3,"n_pointer_only_licence":2}},{"rank_in_archive_order":3,"model":"OFA-DOD-base","metrics":{"Intra-scenario ABS mAP":"15.4","Intra-scenario FULL mAP":"21.6","Intra-scenario PRES mAP":"23.7"},"uses_additional_data":false,"paper_date":"2023-07-24","paper":"/paper/described-object-detection-liberating-object-1","paper_url":"https://arxiv.org/abs/2307.12813v2","paper_title":"Described Object Detection: Liberating Object Detection with Flexible Expressions","code":"https://github.com/shikras/d-cube","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":1,"n_samples":3,"n_pointer_only_licence":3}},{"rank_in_archive_order":4,"model":"GLIP-T","metrics":{"Intra-scenario ABS mAP":"21.5","Intra-scenario FULL mAP":"19.1","Intra-scenario PRES mAP":"18.3"},"uses_additional_data":false,"paper_date":"2021-12-07","paper":"/paper/grounded-language-image-pre-training","paper_url":"https://arxiv.org/abs/2112.03857v2","paper_title":"Grounded Language-Image Pre-training","code":"https://github.com/microsoft/GLIP","n_code_links":3,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":2}},{"rank_in_archive_order":5,"model":"UNINEXT-large","metrics":{"Intra-scenario ABS mAP":"15.9","Intra-scenario FULL mAP":"17.9","Intra-scenario PRES mAP":"18.6"},"uses_additional_data":false,"paper_date":"2023-03-12","paper":"/paper/universal-instance-perception-as-object","paper_url":"https://arxiv.org/abs/2303.06674v2","paper_title":"Universal Instance Perception as Object Discovery and Retrieval","code":"https://github.com/MasterBin-IIAU/UNINEXT","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":1,"n_samples":4,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"SPHINX-7B","metrics":{"Intra-scenario ABS mAP":"7.9","Intra-scenario FULL mAP":"10.6","Intra-scenario PRES mAP":"11.4"},"uses_additional_data":false,"paper_date":"2023-11-13","paper":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and","paper_url":"https://arxiv.org/abs/2311.07575v1","paper_title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","code":"https://github.com/alpha-vllm/llama2-accessory","n_code_links":1,"syntology":{"n_ran":3,"n_unverified":0,"n_samples":3,"n_pointer_only_licence":3}},{"rank_in_archive_order":7,"model":"OWL-ViT-base","metrics":{"Intra-scenario ABS mAP":"8.8","Intra-scenario FULL mAP":"8.6","Intra-scenario PRES mAP":"8.5"},"uses_additional_data":false,"paper_date":"2022-05-12","paper":"/paper/simple-open-vocabulary-object-detection-with","paper_url":"https://arxiv.org/abs/2205.06230v2","paper_title":"Simple Open-Vocabulary Object Detection with Vision Transformers","code":"https://github.com/google-research/scenic/tree/main/scenic/projects/owl_vit","n_code_links":2,"syntology":null},{"rank_in_archive_order":8,"model":"CORA-R50","metrics":{"Intra-scenario ABS mAP":"5.0","Intra-scenario FULL mAP":"6.2","Intra-scenario PRES mAP":"6.7"},"uses_additional_data":false,"paper_date":"2023-03-23","paper":"/paper/cora-adapting-clip-for-open-vocabulary","paper_url":"https://arxiv.org/abs/2303.13076v1","paper_title":"CORA: Adapting CLIP for Open-Vocabulary Detection with Region Prompting and Anchor Pre-Matching","code":"https://github.com/tgxs002/cora","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":1,"n_samples":3,"n_pointer_only_licence":0}}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":7,"rows_with_any_sample_ran":7,"distinct_papers_with_graph_line":7,"distinct_papers_with_any_sample_ran":7,"samples_over_distinct_papers":{"n_ran":18,"n_unverified":6,"n_samples":24,"n_pointer_only_licence":10,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":18,"n_unverified":6,"n_samples":24,"n_pointer_only_licence":10,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}