{"url":"/dataset/mscoco","name":"MSCOCO","full_name":null,"description_markdown":"Click to add a brief description of the dataset (Markdown and LaTeX enabled).\r\n\r\nProvide:\r\n\r\n* a high-level explanation of the dataset characteristics\r\n* explain motivations and summary of its content\r\n* potential use cases of the dataset","description_withheld":null,"homepage":"","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Object Detection","url":"/task/object-detection","datasets_with_task":"/datasets/task/object-detection"},{"name":"Image Retrieval","url":"/task/image-retrieval","datasets_with_task":"/datasets/task/image-retrieval"},{"name":"Image Captioning","url":"/task/image-captioning","datasets_with_task":"/datasets/task/image-captioning"},{"name":"Cross-Modal Retrieval","url":"/task/cross-modal-retrieval","datasets_with_task":"/datasets/task/cross-modal-retrieval"},{"name":"Weakly Supervised Object Detection","url":"/task/weakly-supervised-object-detection","datasets_with_task":"/datasets/task/weakly-supervised-object-detection"},{"name":"Multi-Label Image Classification","url":"/task/multi-label-image-classification","datasets_with_task":"/datasets/task/multi-label-image-classification"},{"name":"Real-time Instance Segmentation","url":"/task/real-time-instance-segmentation","datasets_with_task":"/datasets/task/real-time-instance-segmentation"},{"name":"Zero-Shot Object Detection","url":"/task/zero-shot-object-detection","datasets_with_task":"/datasets/task/zero-shot-object-detection"},{"name":"Open Vocabulary Object Detection","url":"/task/open-vocabulary-object-detection","datasets_with_task":"/datasets/task/open-vocabulary-object-detection"},{"name":"Paraphrase Generation","url":"/task/paraphrase-generation","datasets_with_task":"/datasets/task/paraphrase-generation"},{"name":"Image Outpainting","url":"/task/image-outpainting","datasets_with_task":"/datasets/task/image-outpainting"},{"name":"mage-to-Text Retrieval","url":"/task/mage-to-text-retrieval","datasets_with_task":"/datasets/task/mage-to-text-retrieval"},{"name":"Few Shot Open Set Object Detection","url":"/task/few-shot-open-set-object-detection","datasets_with_task":"/datasets/task/few-shot-open-set-object-detection"}],"languages":[],"variants":[],"data_loaders":[],"num_papers_in_archive":90,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/open-vocabulary-object-detection-on-mscoco","task":"Open Vocabulary Object Detection","dataset_variant":"MSCOCO","rows":32,"metrics":["AP 0.5"],"first_row_in_archive_order":{"model":"Cooperative Foundational Models","paper":"/paper/enhancing-novel-object-detection-via","metrics":{"AP 0.5":"50.3"},"code_links":[{"title":"rohit901/cooperative-foundational-models","url":"https://github.com/rohit901/cooperative-foundational-models"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/real-time-instance-segmentation-on-mscoco","task":"Real-time Instance Segmentation","dataset_variant":"MSCOCO","rows":22,"metrics":["mask AP","AP50","AP75","APS","APM","APL","Frame (fps)"],"first_row_in_archive_order":{"model":"RTMDet-Ins-x","paper":"/paper/rtmdet-an-empirical-study-of-designing-real","metrics":{"AP50":"67.4","AP75":"47.8","APL":"65.5","APS":"22.2","Frame (fps)":"188\n(RTX3090)","mask AP":"44.6"},"code_links":[{"title":"open-mmlab/mmdetection","url":"https://github.com/open-mmlab/mmdetection/tree/3.x/configs/rtmdet"},{"title":"open-mmlab/mmyolo","url":"https://github.com/open-mmlab/mmyolo"},{"title":"open-mmlab/mmrotate","url":"https://github.com/open-mmlab/mmrotate"},{"title":"open-edge-platform/training_extensions","url":"https://github.com/open-edge-platform/training_extensions"},{"title":"PaddlePaddle/PaddleYOLO","url":"https://github.com/PaddlePaddle/PaddleYOLO"},{"title":"open-edge-platform/geti","url":"https://github.com/open-edge-platform/geti"},{"title":"yxb-nku/strip-r-cnn","url":"https://github.com/yxb-nku/strip-r-cnn"},{"title":"HVision-NKU/Strip-R-CNN","url":"https://github.com/HVision-NKU/Strip-R-CNN"},{"title":"fiveai/MoCaE","url":"https://github.com/fiveai/MoCaE"},{"title":"yuyi1005/point2rbox-mmrotate","url":"https://github.com/yuyi1005/point2rbox-mmrotate"},{"title":"cszzshi/SimD","url":"https://github.com/cszzshi/SimD"},{"title":"CycloneBoy/PPDetectionPytorch","url":"https://github.com/CycloneBoy/PPDetectionPytorch"},{"title":"V3Det/mmdetection-V3Det","url":"https://github.com/V3Det/mmdetection-V3Det"},{"title":"RangiLyu/mmdetection_test","url":"https://github.com/RangiLyu/mmdetection_test"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-detection-on-mscoco-6","task":"Object Detection","dataset_variant":"MSCOCO","rows":7,"metrics":["mAP @0.5:0.95","AP"],"first_row_in_archive_order":{"model":"PP-PicoDet-L","paper":"/paper/pp-picodet-a-better-real-time-object-detector","metrics":{"mAP @0.5:0.95":"40.9"},"code_links":[{"title":"PaddlePaddle/PaddleDetection","url":"https://github.com/PaddlePaddle/PaddleDetection"},{"title":"CycloneBoy/PPDetectionPytorch","url":"https://github.com/CycloneBoy/PPDetectionPytorch"},{"title":"umitkacar/ai-edge-computing","url":"https://github.com/umitkacar/ai-edge-computing"},{"title":"developer0hye/PicoDet-Backbone","url":"https://github.com/developer0hye/PicoDet-Backbone"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-object-detection-on-mscoco","task":"Zero-Shot Object Detection","dataset_variant":"MSCOCO","rows":7,"metrics":["AP"],"first_row_in_archive_order":{"model":"Grounding DINO 1.6 Pro (without COCO data)","paper":"/paper/grounding-dino-1-5-advance-the-edge-of-open","metrics":{"AP":"55.4"},"code_links":[{"title":"mit-han-lab/efficientvit","url":"https://github.com/mit-han-lab/efficientvit"},{"title":"idea-research/grounded-sam-2","url":"https://github.com/idea-research/grounded-sam-2"},{"title":"idea-research/grounding-dino-1.5-api","url":"https://github.com/idea-research/grounding-dino-1.5-api"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-label-image-classification-on-mscoco","task":"Multi-Label Image Classification","dataset_variant":"MSCOCO","rows":4,"metrics":["mAP","mean average precision"],"first_row_in_archive_order":{"model":"IDA-R101(H)","paper":"/paper/causality-compensated-attention-for","metrics":{"mAP":"84.8"},"code_links":[{"title":"yu-gi-oh-leilei/IDA_2023ICLR","url":"https://github.com/yu-gi-oh-leilei/IDA_2023ICLR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-retrieval-on-mscoco","task":"Image Retrieval","dataset_variant":"MSCOCO","rows":3,"metrics":["Recall@1","Recall@10","Recall@5"],"first_row_in_archive_order":{"model":"HADA","paper":"/paper/hada-a-graph-based-amalgamation-framework-in","metrics":{"Recall@1":"58.46","Recall@10":"89.66","Recall@5":"82.85"},"code_links":[{"title":"m2man/hada","url":"https://github.com/m2man/hada"},{"title":"m2man/HADA-LAVIS","url":"https://github.com/m2man/HADA-LAVIS"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/cross-modal-retrieval-on-mscoco","task":"Cross-Modal Retrieval","dataset_variant":"MSCOCO","rows":1,"metrics":["Image-to-text R@1"],"first_row_in_archive_order":{"model":"3SHNet","paper":"/paper/3shnet-boosting-image-sentence-retrieval-via","metrics":{"Image-to-text R@1":"85.8"},"code_links":[{"title":"xurige1995/3shnet","url":"https://github.com/xurige1995/3shnet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/few-shot-open-set-object-detection-on-mscoco","task":"Few Shot Open Set Object Detection","dataset_variant":"MSCOCO","rows":1,"metrics":["AR_U"],"first_row_in_archive_order":{"model":"FOODv2","paper":"/paper/hsic-based-moving-weightaveraging-for-few","metrics":{"AR_U":"16.52"},"code_links":[{"title":"binyisu/food","url":"https://github.com/binyisu/food"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-captioning-on-mscoco-1","task":"Image Captioning","dataset_variant":"MSCOCO","rows":1,"metrics":["BLEU-4"],"first_row_in_archive_order":{"model":"CapDec","paper":"/paper/text-only-training-for-image-captioning-using","metrics":{"BLEU-4":"26.4"},"code_links":[{"title":"davidhuji/capdec","url":"https://github.com/davidhuji/capdec"},{"title":"zelaki/wsac","url":"https://github.com/zelaki/wsac"},{"title":"avitej-iyer/CapDec-Recreation","url":"https://github.com/avitej-iyer/CapDec-Recreation"},{"title":"uriberger/re_cap","url":"https://github.com/uriberger/re_cap"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-outpainting-on-mscoco","task":"Image Outpainting","dataset_variant":"MSCOCO","rows":1,"metrics":["CLIP Similarity","FID","Inception score"],"first_row_in_archive_order":{"model":"NUWA-3D","paper":"/paper/learning-3d-photography-videos-via-self","metrics":{"CLIP Similarity":"32.26","FID":"10.65","Inception score":"38.61"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-detection-on-mscoco-7","task":"Object Detection","dataset_variant":"MSCOCO","rows":1,"metrics":["Average mAP"],"first_row_in_archive_order":{"model":"DAS","paper":"/paper/das-a-deformable-attention-to-capture-salient","metrics":{"Average mAP":"39.7"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/paraphrase-generation-on-mscoco","task":"Paraphrase Generation","dataset_variant":"MSCOCO","rows":1,"metrics":["BLEU","iBLEU"],"first_row_in_archive_order":{"model":"HRQ-VAE","paper":"/paper/hierarchical-sketch-induction-for-paraphrase","metrics":{"BLEU":"27.90","iBLEU":"19.04"},"code_links":[{"title":"tomhosking/hrq-vae","url":"https://github.com/tomhosking/hrq-vae"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/weakly-supervised-object-detection-on-mscoco","task":"Weakly Supervised Object Detection","dataset_variant":"MSCOCO","rows":1,"metrics":["mAP","mAP@50"],"first_row_in_archive_order":{"model":"CASD(ResNet50)","paper":"/paper/comprehensive-attention-self-distillation-for","metrics":{"mAP":"13.9","mAP@50":"27.8"},"code_links":[{"title":"DeLightCMU/CASD","url":"https://github.com/DeLightCMU/CASD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/cp-detr-concept-prompt-guide-detr-toward","title":"CP-DETR: Concept Prompt Guide DETR Toward Stronger Universal Object Detection","date":"2024-12-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/sia-ovd-shape-invariant-adapter-for-bridging","title":"SIA-OVD: Shape-Invariant Adapter for Bridging the Image-Region Gap in Open-Vocabulary Detection","date":"2024-10-08","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/ov-dino-unified-open-vocabulary-detection","title":"OV-DINO: Unified Open-Vocabulary Detection with Language-Aware Selective Fusion","date":"2024-07-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/ov-dquo-open-vocabulary-detr-with-denoising","title":"OV-DQUO: Open-Vocabulary DETR with Denoising Text Query Training and Open-World Unknown Objects Supervision","date":"2024-05-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":9,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/grounding-dino-1-5-advance-the-edge-of-open","title":"Grounding DINO 1.5: Advance the \"Edge\" of Open-Set Object Detection","date":"2024-05-16","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/3shnet-boosting-image-sentence-retrieval-via","title":"3SHNet: Boosting Image-Sentence Retrieval via Visual Semantic-Spatial Self-Highlighting","date":"2024-04-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":10,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/retrieval-augmented-open-vocabulary-object","title":"Retrieval-Augmented Open-Vocabulary Object Detection","date":"2024-04-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/yolov8-am-yolov8-with-attention-mechanisms","title":"YOLOv8-AM: YOLOv8 Based on Effective Attention Mechanisms for Pediatric Wrist Fracture Detection","date":"2024-02-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/yolo-world-real-time-open-vocabulary-object","title":"YOLO-World: Real-Time Open-Vocabulary Object Detection","date":"2024-01-30","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/clim-contrastive-language-image-mosaic-for","title":"CLIM: Contrastive Language-Image Mosaic for Region Representation","date":"2023-12-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/das-a-deformable-attention-to-capture-salient","title":"DAS: A Deformable Attention to Capture Salient Information in CNNs","date":"2023-11-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/enhancing-novel-object-detection-via","title":"Enhancing Novel Object Detection via Cooperative Foundational Models","date":"2023-11-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/yolov8-based-visual-detection-of-road-hazards","title":"YOLOv8-Based Visual Detection of Road Hazards: Potholes, Sewer Covers, and Manholes","date":"2023-10-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hsic-based-moving-weightaveraging-for-few","title":"HSIC-based Moving WeightAveraging for Few-Shot Open-Set Object Detection","date":"2023-10-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/lp-ovod-open-vocabulary-object-detection-by","title":"LP-OVOD: Open-Vocabulary Object Detection by Linear Probing","date":"2023-10-26","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/clipself-vision-transformer-distills-itself","title":"CLIPSelf: Vision Transformer Distills Itself for Open-Vocabulary Dense Prediction","date":"2023-10-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/detection-oriented-image-text-pretraining-for","title":"Region-centric Image-Language Pretraining for Open-Vocabulary Detection","date":"2023-09-29","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/detect-every-thing-with-few-examples","title":"Detect Everything with Few Examples","date":"2023-09-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":15,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/contrastive-feature-masking-open-vocabulary","title":"Contrastive Feature Masking Open-Vocabulary Vision Transformer","date":"2023-09-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/scaledet-a-scalable-multi-dataset-object-1","title":"ScaleDet: A Scalable Multi-Dataset Object Detector","date":"2023-06-08","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cora-adapting-clip-for-open-vocabulary","title":"CORA: Adapting CLIP for Open-Vocabulary Detection with Region Prompting and Anchor Pre-Matching","date":"2023-03-23","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/object-aware-distillation-pyramid-for-open","title":"Object-Aware Distillation Pyramid for Open-Vocabulary Object Detection","date":"2023-03-10","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/grounding-dino-marrying-dino-with-grounded","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","date":"2023-03-09","rows_on_this_dataset":1,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/aligning-bag-of-regions-for-open-vocabulary","title":"Aligning Bag of Regions for Open-Vocabulary Object Detection","date":"2023-02-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/causality-compensated-attention-for","title":"Causality Compensated Attention for Contextual Biased Visual Recognition","date":"2023-02-25","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/learning-3d-photography-videos-via-self","title":"Learning 3D Photography Videos via Self-supervised Diffusion on Single Images","date":"2023-02-21","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hada-a-graph-based-amalgamation-framework-in","title":"HADA: A Graph-based Amalgamation Framework in Image-text Retrieval","date":"2023-01-11","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/rtmdet-an-empirical-study-of-designing-real","title":"RTMDet: An Empirical Study of Designing Real-Time Object Detectors","date":"2022-12-14","rows_on_this_dataset":4,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":3,"samples_unverified":17,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/open-vocabulary-attribute-detection","title":"Open-vocabulary Attribute Detection","date":"2022-11-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":3,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/text-only-training-for-image-captioning-using","title":"Text-Only Training for Image Captioning using Noise-Injected CLIP","date":"2022-11-01","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-label-image-classification-using-1","title":"Multi Label Image Classification using Adaptive Graph Convolutional Networks (ML-AGCN)","date":"2022-10-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/exploiting-unlabeled-data-with-vision-and","title":"Exploiting Unlabeled Data with Vision and Language Models for Object Detection","date":"2022-07-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/bridging-the-gap-between-object-and-image","title":"Bridging the Gap between Object and Image-level Representations for Open-Vocabulary Detection","date":"2022-07-07","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/open-vocabulary-object-detection-with","title":"Open Vocabulary Object Detection with Proposal Mining and Prediction Equalization","date":"2022-06-22","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":7,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/localized-vision-language-matching-for-open","title":"Localized Vision-Language Matching for Open-vocabulary Object Detection","date":"2022-05-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/sparse-instance-activation-for-real-time","title":"Sparse Instance Activation for Real-Time Instance Segmentation","date":"2022-03-24","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/open-vocabulary-detr-with-conditional","title":"Open-Vocabulary DETR with Conditional Matching","date":"2022-03-22","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":3,"samples_unverified":5,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/open-vocabulary-one-stage-detection-with","title":"Open-Vocabulary One-Stage Detection with Hierarchical Visual-Language Knowledge Distillation","date":"2022-03-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hierarchical-sketch-induction-for-paraphrase","title":"Hierarchical Sketch Induction for Paraphrase Generation","date":"2022-03-07","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/detecting-twenty-thousand-classes-using-image","title":"Detecting Twenty-thousand Classes using Image-level Supervision","date":"2022-01-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/regionclip-region-based-language-image","title":"RegionCLIP: Region-based Language-Image Pretraining","date":"2021-12-16","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/pp-picodet-a-better-real-time-object-detector","title":"PP-PicoDet: A Better Real-Time Object Detector on Mobile Devices","date":"2021-11-01","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/mask-aware-iou-for-anchor-assignment-in-real","title":"Mask-aware IoU for Anchor Assignment in Real-time Instance Segmentation","date":"2021-10-19","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/zero-shot-detection-via-vision-and-language","title":"Open-vocabulary Object Detection via Vision and Language Knowledge Distillation","date":"2021-04-28","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/open-vocabulary-object-detection-using","title":"Open-Vocabulary Object Detection Using Captions","date":"2020-11-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/comprehensive-attention-self-distillation-for","title":"Comprehensive Attention Self-Distillation for Weakly-Supervised Object Detection","date":"2020-10-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sipmask-spatial-information-preservation-for","title":"SipMask: Spatial Information Preservation for Fast Image and Video Instance Segmentation","date":"2020-07-29","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/solov2-dynamic-faster-and-stronger","title":"SOLOv2: Dynamic and Fast Instance Segmentation","date":"2020-03-23","rows_on_this_dataset":1,"code_links":18,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":38,"samples_ran":15,"samples_unverified":23,"pointer_only_for_licence":24,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/blendmask-top-down-meets-bottom-up-for","title":"BlendMask: Top-Down Meets Bottom-Up for Instance Segmentation","date":"2020-01-02","rows_on_this_dataset":1,"code_links":9,"syntology":null},{"paper":"/paper/yolact-better-real-time-instance-segmentation","title":"YOLACT++: Better Real-time Instance Segmentation","date":"2019-12-03","rows_on_this_dataset":1,"code_links":36,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":43,"samples_ran":11,"samples_unverified":32,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/centermask-real-time-anchor-free-instance-1","title":"CenterMask : Real-Time Anchor-Free Instance Segmentation","date":"2019-11-15","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/yolact-real-time-instance-segmentation","title":"YOLACT: Real-time Instance Segmentation","date":"2019-04-04","rows_on_this_dataset":2,"code_links":48,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":10,"samples_unverified":11,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":24,"samples_harvested":248,"samples_ran":114,"samples_unverified":134,"pointer_only_for_licence":54,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}