{"url":"/dataset/coco","name":"COCO (Common Objects in Context)","full_name":"Common Objects in Context","description_markdown":"The COCO (Common Objects in Context) dataset is a large-scale object detection, segmentation, and captioning dataset. It is designed to encourage research on a wide variety of object categories and is commonly used for benchmarking computer vision models. It is an essential dataset for researchers and developers working on object detection, segmentation, and pose estimation tasks.","description_withheld":null,"homepage":"","introduced_date":"2014-01-01","introduced_date_note":null,"introduced_by":null,"license":{"name":"Custom","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Videos","url":"/datasets/modality/videos"}],"tasks":[{"name":"Object Detection","url":"/task/object-detection","datasets_with_task":"/datasets/task/object-detection"},{"name":"Semantic Segmentation","url":"/task/semantic-segmentation","datasets_with_task":"/datasets/task/semantic-segmentation"},{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Pose Estimation","url":"/task/pose-estimation","datasets_with_task":"/datasets/task/pose-estimation"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"},{"name":"Instance Segmentation","url":"/task/instance-segmentation","datasets_with_task":"/datasets/task/instance-segmentation"},{"name":"Image Retrieval","url":"/task/image-retrieval","datasets_with_task":"/datasets/task/image-retrieval"},{"name":"Visual Question Answering","url":"/task/visual-question-answering-1","datasets_with_task":"/datasets/task/visual-question-answering-1"},{"name":"Image Captioning","url":"/task/image-captioning","datasets_with_task":"/datasets/task/image-captioning"},{"name":"Multi-Label Classification","url":"/task/multi-label-classification","datasets_with_task":"/datasets/task/multi-label-classification"},{"name":"Object Localization","url":"/task/object-localization","datasets_with_task":"/datasets/task/object-localization"},{"name":"Panoptic Segmentation","url":"/task/panoptic-segmentation","datasets_with_task":"/datasets/task/panoptic-segmentation"},{"name":"Unsupervised Semantic Segmentation","url":"/task/unsupervised-semantic-segmentation","datasets_with_task":"/datasets/task/unsupervised-semantic-segmentation"},{"name":"Text-to-Image Generation","url":"/task/text-to-image-generation","datasets_with_task":"/datasets/task/text-to-image-generation"},{"name":"Cross-Modal Retrieval","url":"/task/cross-modal-retrieval","datasets_with_task":"/datasets/task/cross-modal-retrieval"},{"name":"Weakly Supervised Object Detection","url":"/task/weakly-supervised-object-detection","datasets_with_task":"/datasets/task/weakly-supervised-object-detection"},{"name":"Few-Shot Object Detection","url":"/task/few-shot-object-detection","datasets_with_task":"/datasets/task/few-shot-object-detection"},{"name":"Multi-Person Pose Estimation","url":"/task/multi-person-pose-estimation","datasets_with_task":"/datasets/task/multi-person-pose-estimation"},{"name":"Object Counting","url":"/task/object-counting","datasets_with_task":"/datasets/task/object-counting"},{"name":"Conditional Image Generation","url":"/task/conditional-image-generation","datasets_with_task":"/datasets/task/conditional-image-generation"},{"name":"Interactive Segmentation","url":"/task/interactive-segmentation","datasets_with_task":"/datasets/task/interactive-segmentation"},{"name":"Quantization","url":"/task/quantization","datasets_with_task":"/datasets/task/quantization"},{"name":"Question Generation","url":"/task/question-generation","datasets_with_task":"/datasets/task/question-generation"},{"name":"Unsupervised Semantic Segmentation with Language-image Pre-training","url":"/task/unsupervised-semantic-segmentation-with","datasets_with_task":"/datasets/task/unsupervised-semantic-segmentation-with"},{"name":"Zero-Shot Composed Image Retrieval (ZS-CIR)","url":"/task/zero-shot-composed-image-retrieval-zs-cir","datasets_with_task":"/datasets/task/zero-shot-composed-image-retrieval-zs-cir"},{"name":"Layout-to-Image Generation","url":"/task/layout-to-image-generation","datasets_with_task":"/datasets/task/layout-to-image-generation"},{"name":"Knowledge Distillation","url":"/task/knowledge-distillation","datasets_with_task":"/datasets/task/knowledge-distillation"},{"name":"Multi-Label Image Classification","url":"/task/multi-label-image-classification","datasets_with_task":"/datasets/task/multi-label-image-classification"},{"name":"Scene Graph Generation","url":"/task/scene-graph-generation","datasets_with_task":"/datasets/task/scene-graph-generation"},{"name":"Keypoint Detection","url":"/task/keypoint-detection","datasets_with_task":"/datasets/task/keypoint-detection"},{"name":"Real-Time Object Detection","url":"/task/real-time-object-detection","datasets_with_task":"/datasets/task/real-time-object-detection"},{"name":"Real-time Instance Segmentation","url":"/task/real-time-instance-segmentation","datasets_with_task":"/datasets/task/real-time-instance-segmentation"},{"name":"Image-to-Text Retrieval","url":"/task/image-to-text-retrieval","datasets_with_task":"/datasets/task/image-to-text-retrieval"},{"name":"Zero-Shot Object Detection","url":"/task/zero-shot-object-detection","datasets_with_task":"/datasets/task/zero-shot-object-detection"},{"name":"Open World Object Detection","url":"/task/open-world-object-detection","datasets_with_task":"/datasets/task/open-world-object-detection"},{"name":"Open Vocabulary Object Detection","url":"/task/open-vocabulary-object-detection","datasets_with_task":"/datasets/task/open-vocabulary-object-detection"},{"name":"Homography Estimation","url":"/task/homography-estimation","datasets_with_task":"/datasets/task/homography-estimation"},{"name":"Robust Object Detection","url":"/task/robust-object-detection","datasets_with_task":"/datasets/task/robust-object-detection"},{"name":"Single-object discovery","url":"/task/single-object-discovery","datasets_with_task":"/datasets/task/single-object-discovery"},{"name":"Zero-shot Text-to-Image Retrieval","url":"/task/zero-shot-text-to-image-retrieval","datasets_with_task":"/datasets/task/zero-shot-text-to-image-retrieval"},{"name":"Paraphrase Generation","url":"/task/paraphrase-generation","datasets_with_task":"/datasets/task/paraphrase-generation"},{"name":"Unsupervised Object Localization","url":"/task/unsupervised-object-localization","datasets_with_task":"/datasets/task/unsupervised-object-localization"},{"name":"Weakly-supervised instance segmentation","url":"/task/weakly-supervised-instance-segmentation","datasets_with_task":"/datasets/task/weakly-supervised-instance-segmentation"},{"name":"Image Outpainting","url":"/task/image-outpainting","datasets_with_task":"/datasets/task/image-outpainting"},{"name":"Multi-object discovery","url":"/task/multi-object-discovery","datasets_with_task":"/datasets/task/multi-object-discovery"},{"name":"Zero-Shot Cross-Modal Retrieval","url":"/task/zero-shot-cross-modal-retrieval","datasets_with_task":"/datasets/task/zero-shot-cross-modal-retrieval"},{"name":"Semi Supervised Learning for Image Captioning","url":"/task/semi-supervised-learning-for-image-captioning","datasets_with_task":"/datasets/task/semi-supervised-learning-for-image-captioning"},{"name":"Image-level Supervised Instance Segmentation","url":"/task/image-level-supervised-instance-segmentation","datasets_with_task":"/datasets/task/image-level-supervised-instance-segmentation"},{"name":"Point-Supervised Instance Segmentation","url":"/task/point-supervised-instance-segmentation","datasets_with_task":"/datasets/task/point-supervised-instance-segmentation"},{"name":"Object Proposal Generation","url":"/task/object-proposal-generation","datasets_with_task":"/datasets/task/object-proposal-generation"},{"name":"One-Shot Object Detection","url":"/task/one-shot-object-detection","datasets_with_task":"/datasets/task/one-shot-object-detection"},{"name":"Active Object Detection","url":"/task/active-object-detection","datasets_with_task":"/datasets/task/active-object-detection"},{"name":"Box-supervised Instance Segmentation","url":"/task/box-supervised-instance-segmentation","datasets_with_task":"/datasets/task/box-supervised-instance-segmentation"},{"name":"mage-to-Text Retrieval","url":"/task/mage-to-text-retrieval","datasets_with_task":"/datasets/task/mage-to-text-retrieval"},{"name":"Multi-Label Learning","url":"/task/multi-label-learning","datasets_with_task":"/datasets/task/multi-label-learning"},{"name":"One-Shot Instance Segmentation","url":"/task/one-shot-instance-segmentation","datasets_with_task":"/datasets/task/one-shot-instance-segmentation"},{"name":"Region Proposal","url":"/task/region-proposal","datasets_with_task":"/datasets/task/region-proposal"},{"name":"Activeness Detection","url":"/task/activeness-detection","datasets_with_task":"/datasets/task/activeness-detection"},{"name":"Generalized Zero-Shot Object Detection","url":"/task/generalized-zero-shot-object-detection","datasets_with_task":"/datasets/task/generalized-zero-shot-object-detection"},{"name":"Few Shot Open Set Object Detection","url":"/task/few-shot-open-set-object-detection","datasets_with_task":"/datasets/task/few-shot-open-set-object-detection"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["MSCOCO-1k","COCO 2017 (Sports, Food)","COCO 2017 (Outdoor, Accessories, Appliance, Truck)","COCO 2017 (Electronic, Indoor, Kitchen, Furniture)","COCO Visual Question Answering (VQA) abstract images 1.0 open ended","COCO Visual Question Answering (VQA) abstract 1.0 multiple choice","COCO 2017","COCO 2014","coco minval","COCO-Stuff-3","COCO-Stuff 256x256","COCO 256 x 256","COCO 2015","MS-COCO (10-shot)","MS-COCO","COCO Visual Question Answering (VQA) real images 2.0 open ended","COCO Visual Question Answering (VQA) real images 1.0 open ended","MSCOCO","DensePose-COCO","COCO_20k","COCO-Animals","COCO+","COCO test-dev","COCO test-challenge","COCO panoptic","COCO minival","COCO count-test","COCO Visual Question Answering (VQA) real images 1.0 multiple choice","COCO (image as query)","COCO (Common Objects in Context)"],"data_loaders":[{"repo":"https://github.com/facebookresearch/detectron2","url":"https://detectron2.readthedocs.io/en/latest/tutorials/builtin_datasets.html#expected-dataset-structure-for-coco-instance-keypoint-detection","frameworks":["pytorch"]},{"repo":"https://github.com/open-mmlab/mmdetection","url":"https://github.com/open-mmlab/mmdetection/blob/master/docs/1_exist_data_model.md","frameworks":["pytorch"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/coallaoh/COCO-AB","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/pytorch/vision","url":"https://pytorch.org/vision/stable/datasets.html#torchvision.datasets.CocoDetection","frameworks":["pytorch"]},{"repo":"https://github.com/PaddlePaddle/PaddleDetection","url":"https://github.com/PaddlePaddle/PaddleDetection","frameworks":[]},{"repo":"https://github.com/voxel51/fiftyone","url":"https://docs.voxel51.com/user_guide/dataset_zoo/datasets.html#dataset-zoo-coco-2017","frameworks":[]},{"repo":"https://github.com/activeloopai/Hub","url":"https://docs.activeloop.ai/datasets/coco-dataset","frameworks":["tf","pytorch"]},{"repo":"https://github.com/open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose/blob/master/docs/tasks/2d_body_keypoint.md#coco","frameworks":["pytorch"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/coco","frameworks":["tf","jax"]}],"num_papers_in_archive":11922,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/object-detection-on-coco","task":"Object Detection","dataset_variant":"COCO test-dev","rows":225,"metrics":["box mAP","AP50","AP75","APS","APM","APL","Hardware Burden","Params (M)","Operations per network pass","GFLOPs"],"first_row_in_archive_order":{"model":"Co-DETR","paper":"/paper/detrs-with-collaborative-hybrid-assignments","metrics":{"Params (M)":"304","box mAP":"66.0"},"code_links":[{"title":"open-mmlab/mmdetection","url":"https://github.com/open-mmlab/mmdetection"},{"title":"siyuanliii/masa","url":"https://github.com/siyuanliii/masa"},{"title":"sense-x/co-detr","url":"https://github.com/sense-x/co-detr"},{"title":"anenbergb/Co-DETR-TensorRT","url":"https://github.com/anenbergb/Co-DETR-TensorRT"},{"title":"MindCode-4/code-3","url":"https://github.com/MindCode-4/code-3/tree/main/detr"},{"title":"code-implementation1/Code1","url":"https://github.com/code-implementation1/Code1/tree/main/detr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-detection-on-coco-minival","task":"Object Detection","dataset_variant":"COCO minival","rows":220,"metrics":["box AP","AP50","AP75","APS","APM","APL","Params (M)"],"first_row_in_archive_order":{"model":"PE_spatial (DETA)","paper":"/paper/perception-encoder-the-best-visual-embeddings","metrics":{"Params (M)":"1900","box AP":"66.0"},"code_links":[{"title":"facebookresearch/perception_models","url":"https://github.com/facebookresearch/perception_models"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/instance-segmentation-on-coco","task":"Instance Segmentation","dataset_variant":"COCO test-dev","rows":112,"metrics":["mask AP","AP50","AP75","APS","APM","APL"],"first_row_in_archive_order":{"model":"Co-DETR","paper":"/paper/detrs-with-collaborative-hybrid-assignments","metrics":{"AP50":"80.2","AP75":"63.4","APL":"72.0","APM":"60.1","APS":"41.6","mask AP":"57.1"},"code_links":[{"title":"open-mmlab/mmdetection","url":"https://github.com/open-mmlab/mmdetection"},{"title":"siyuanliii/masa","url":"https://github.com/siyuanliii/masa"},{"title":"sense-x/co-detr","url":"https://github.com/sense-x/co-detr"},{"title":"anenbergb/Co-DETR-TensorRT","url":"https://github.com/anenbergb/Co-DETR-TensorRT"},{"title":"MindCode-4/code-3","url":"https://github.com/MindCode-4/code-3/tree/main/detr"},{"title":"code-implementation1/Code1","url":"https://github.com/code-implementation1/Code1/tree/main/detr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/instance-segmentation-on-coco-minival","task":"Instance Segmentation","dataset_variant":"COCO minival","rows":93,"metrics":["mask AP","AP50","AP75","APL","APM","APS","GFLOPs","Params (M)","box AP"],"first_row_in_archive_order":{"model":"Co-DETR","paper":"/paper/detrs-with-collaborative-hybrid-assignments","metrics":{"AP50":"79.7","AP75":"62.8","APL":"74.6","APM":"59.7","APS":"38.9","mask AP":"56.6"},"code_links":[{"title":"open-mmlab/mmdetection","url":"https://github.com/open-mmlab/mmdetection"},{"title":"siyuanliii/masa","url":"https://github.com/siyuanliii/masa"},{"title":"sense-x/co-detr","url":"https://github.com/sense-x/co-detr"},{"title":"anenbergb/Co-DETR-TensorRT","url":"https://github.com/anenbergb/Co-DETR-TensorRT"},{"title":"MindCode-4/code-3","url":"https://github.com/MindCode-4/code-3/tree/main/detr"},{"title":"code-implementation1/Code1","url":"https://github.com/code-implementation1/Code1/tree/main/detr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/real-time-object-detection-on-coco","task":"Real-Time Object Detection","dataset_variant":"COCO (Common Objects in Context)","rows":82,"metrics":["box AP","FPS (V100, b=1)"],"first_row_in_archive_order":{"model":"DEIM-D-FINE-X+","paper":"/paper/deim-detr-with-improved-matching-for-fast","metrics":{"FPS (V100, b=1)":"78 (T4)","box AP":"59.5"},"code_links":[{"title":"shihuahuang95/deim","url":"https://github.com/shihuahuang95/deim"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-to-image-generation-on-coco","task":"Text-to-Image Generation","dataset_variant":"COCO (Common Objects in Context)","rows":69,"metrics":["FID","Inception score","FID-1","FID-2","FID-4","FID-8","SOA-C","Zero shot FID"],"first_row_in_archive_order":{"model":"RAT-Diffusion","paper":"/paper/data-extrapolation-for-text-to-image","metrics":{"FID":"5.00"},"code_links":[{"title":"senmaoy/RAT-Diffusion","url":"https://github.com/senmaoy/RAT-Diffusion"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/pose-estimation-on-coco-test-dev","task":"Pose Estimation","dataset_variant":"COCO test-dev","rows":47,"metrics":["AP","AP50","AP75","APL","APM","AR"],"first_row_in_archive_order":{"model":"ViTPose (ViTAE-G, ensemble)","paper":"/paper/vitpose-simple-vision-transformer-baselines","metrics":{"AP":"81.1","AP50":"95.0","AP75":"88.2","APL":"86.0","APM":"77.8","AR":"85.6"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"vitae-transformer/vitpose","url":"https://github.com/vitae-transformer/vitpose"},{"title":"vitae-transformer/qformer","url":"https://github.com/vitae-transformer/qformer"},{"title":"JunkyByte/easy_ViTPose","url":"https://github.com/JunkyByte/easy_ViTPose"},{"title":"jaehyunnn/ViTPose_pytorch","url":"https://github.com/jaehyunnn/ViTPose_pytorch"},{"title":"gpastal24/ViTPose-Pytorch","url":"https://github.com/gpastal24/ViTPose-Pytorch"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/panoptic-segmentation-on-coco-test-dev","task":"Panoptic Segmentation","dataset_variant":"COCO test-dev","rows":38,"metrics":["PQ","PQst","PQth"],"first_row_in_archive_order":{"model":"Mask DINO (single scale)","paper":"/paper/mask-dino-towards-a-unified-transformer-based-1","metrics":{"PQ":"59.5","PQst":"-","PQth":"-"},"code_links":[{"title":"PaddlePaddle/PaddleDetection","url":"https://github.com/PaddlePaddle/PaddleDetection"},{"title":"IDEACVR/DINO","url":"https://github.com/IDEACVR/DINO"},{"title":"idea-research/maskdino","url":"https://github.com/idea-research/maskdino"},{"title":"idea-research/dn-detr","url":"https://github.com/idea-research/dn-detr"},{"title":"IDEA-opensource/DN-DETR","url":"https://github.com/IDEA-opensource/DN-DETR"},{"title":"IDEA-opensource/DAB-DETR","url":"https://github.com/IDEA-opensource/DAB-DETR"},{"title":"idea-research/dab-detr","url":"https://github.com/idea-research/dab-detr"},{"title":"isbrycee/gem-glass-segmentor","url":"https://github.com/isbrycee/gem-glass-segmentor"},{"title":"isbrycee/gem","url":"https://github.com/isbrycee/gem"},{"title":"Expedit-LargeScale-Vision-Transformer/Expedit-DINO","url":"https://github.com/Expedit-LargeScale-Vision-Transformer/Expedit-DINO"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/cross-modal-retrieval-on-coco-2014","task":"Cross-Modal Retrieval","dataset_variant":"COCO 2014","rows":36,"metrics":["Text-to-image R@1","Text-to-image R@5","Text-to-image R@10","Image-to-text R@1","Image-to-text R@5","Image-to-text R@10"],"first_row_in_archive_order":{"model":"VAST","paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","metrics":{"Text-to-image R@1":"68.0","Text-to-image R@10":"92.8","Text-to-image R@5":"87.7"},"code_links":[{"title":"TXH-mercury/VALOR","url":"https://github.com/TXH-mercury/VALOR"},{"title":"txh-mercury/vast","url":"https://github.com/txh-mercury/vast"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-label-classification-on-ms-coco","task":"Multi-Label Classification","dataset_variant":"MS-COCO","rows":34,"metrics":["mAP"],"first_row_in_archive_order":{"model":"ADDS(ViT-L-336, resolution 1344)","paper":"/paper/a-dual-modality-approach-for-zero-shot-multi","metrics":{"mAP":"93.54"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/few-shot-object-detection-on-ms-coco-10-shot","task":"Few-Shot Object Detection","dataset_variant":"MS-COCO (10-shot)","rows":33,"metrics":["AP"],"first_row_in_archive_order":{"model":"Training-free","paper":"/paper/no-time-to-train-training-free-reference","metrics":{"AP":"36.6"},"code_links":[{"title":"miquel-espinosa/no-time-to-train","url":"https://github.com/miquel-espinosa/no-time-to-train"},{"title":"miquel-espinosa/samantics","url":"https://github.com/miquel-espinosa/samantics"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/panoptic-segmentation-on-coco-minival","task":"Panoptic Segmentation","dataset_variant":"COCO minival","rows":31,"metrics":["PQ","PQst","PQth","RQ","SQ","RQst","RQth","SQst","SQth","AP","mIoU","boxAP","maskAP"],"first_row_in_archive_order":{"model":"HyperSeg (Swin-B)","paper":"/paper/hyperseg-towards-universal-visual","metrics":{"PQ":"61.2"},"code_links":[{"title":"congvvc/HyperSeg","url":"https://github.com/congvvc/HyperSeg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/keypoint-detection-on-coco","task":"Keypoint Detection","dataset_variant":"COCO (Common Objects in Context)","rows":24,"metrics":["Test AP","Validation AP","FPS"],"first_row_in_archive_order":{"model":"4xRSN-50(384×288)","paper":"/paper/learning-delicate-local-representations-for","metrics":{"Test AP":"78.6"},"code_links":[{"title":"open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose"},{"title":"chenyilun95/tf-cpn","url":"https://github.com/chenyilun95/tf-cpn"},{"title":"caiyuanhao1998/RSN","url":"https://github.com/caiyuanhao1998/RSN"},{"title":"HuangJunJie2017/UDP-Pose","url":"https://github.com/HuangJunJie2017/UDP-Pose"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-detection-on-coco-2017","task":"Object Detection","dataset_variant":"COCO 2017","rows":24,"metrics":["AP","mAP","Mean mAP","AP50","AP75","APM","APM50","APM75"],"first_row_in_archive_order":{"model":"MaxViT-B","paper":"/paper/maxvit-multi-axis-vision-transformer","metrics":{"AP":"53.4","AP50":"72.9","AP75":"58.1","APM":"45.7","APM50":"70.3","APM75":"50"},"code_links":[{"title":"huggingface/pytorch-image-models","url":"https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/maxxvit.py"},{"title":"lucidrains/vit-pytorch","url":"https://github.com/lucidrains/vit-pytorch"},{"title":"lucidrains/imagen-pytorch","url":"https://github.com/lucidrains/imagen-pytorch"},{"title":"towhee-io/towhee","url":"https://github.com/towhee-io/towhee"},{"title":"google-research/maxim","url":"https://github.com/google-research/maxim"},{"title":"leondgarse/keras_cv_attention_models","url":"https://github.com/leondgarse/keras_cv_attention_models/tree/main/keras_cv_attention_models/maxvit"},{"title":"google-research/maxvit","url":"https://github.com/google-research/maxvit"},{"title":"ChristophReich1996/MaxViT","url":"https://github.com/ChristophReich1996/MaxViT"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"qwopqwop200/MaxVIT-pytorch","url":"https://github.com/qwopqwop200/MaxVIT-pytorch"},{"title":"RooKichenn/pytorch-MaxViT","url":"https://github.com/RooKichenn/pytorch-MaxViT"},{"title":"hankyul2/maxvit-pytorch","url":"https://github.com/hankyul2/maxvit-pytorch"},{"title":"2024-MindSpore-1/Code3","url":"https://github.com/2024-MindSpore-1/Code3/tree/main/MaxViT"},{"title":"Mind23-2/MindCode-3","url":"https://github.com/Mind23-2/MindCode-3/tree/main/NFNet"},{"title":"Mind23-2/MindCode-101","url":"https://github.com/Mind23-2/MindCode-101/tree/main/NFNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-cross-modal-retrieval-on-coco-2014","task":"Zero-Shot Cross-Modal Retrieval","dataset_variant":"COCO 2014","rows":18,"metrics":["Image-to-text R@1","Image-to-text R@5","Image-to-text R@10","Text-to-image R@1","Text-to-image R@5","Text-to-image R@10"],"first_row_in_archive_order":{"model":"InternVL-G","paper":"/paper/internvl-scaling-up-vision-foundation-models","metrics":{"Image-to-text R@1":"74.9","Image-to-text R@10":"95.2","Image-to-text R@5":"91.3","Text-to-image R@1":"58.6","Text-to-image R@10":"88.0","Text-to-image R@5":"81.3"},"code_links":[{"title":"opengvlab/internvl","url":"https://github.com/opengvlab/internvl"},{"title":"opengvlab/internvl-mmdetseg","url":"https://github.com/opengvlab/internvl-mmdetseg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-captioning-on-coco","task":"Image Captioning","dataset_variant":"COCO (Common Objects in Context)","rows":17,"metrics":["CIDEr","BLEU-1","BLEU-2","BLEU-3","BLEU-4","METEOR","ROUGE","ROUGE-L"],"first_row_in_archive_order":{"model":"ExpansionNet v2","paper":"/paper/expansionnet-v2-block-static-expansion-in","metrics":{"CIDEr":"143.7"},"code_links":[{"title":"jchenghu/expansionnet_v2","url":"https://github.com/jchenghu/expansionnet_v2"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/keypoint-detection-on-coco-test-dev","task":"Keypoint Detection","dataset_variant":"COCO test-dev","rows":16,"metrics":["APM","APL","AP50","AP75","AR","AR50","AR75","ARM","ARL","AP"],"first_row_in_archive_order":{"model":"HRNet*","paper":"/paper/deep-high-resolution-representation-learning","metrics":{"AP50":"92.7","AP75":"84.5","APL":"83.1","APM":"73.4","AR":"82.0"},"code_links":[{"title":"open-mmlab/mmdetection","url":"https://github.com/open-mmlab/mmdetection"},{"title":"PaddlePaddle/PaddleDetection","url":"https://github.com/PaddlePaddle/PaddleDetection"},{"title":"open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose"},{"title":"leoxiaobin/deep-high-resolution-net.pytorch","url":"https://github.com/leoxiaobin/deep-high-resolution-net.pytorch"},{"title":"HRNet/HRNet-Semantic-Segmentation","url":"https://github.com/HRNet/HRNet-Semantic-Segmentation"},{"title":"osmr/imgclsmob","url":"https://github.com/osmr/imgclsmob"},{"title":"Microsoft/human-pose-estimation.pytorch","url":"https://github.com/Microsoft/human-pose-estimation.pytorch"},{"title":"HRNet/HRNet-Facial-Landmark-Detection","url":"https://github.com/HRNet/HRNet-Facial-Landmark-Detection"},{"title":"HRNet/HRNet-Image-Classification","url":"https://github.com/HRNet/HRNet-Image-Classification"},{"title":"HRNet/HRNet-Object-Detection","url":"https://github.com/HRNet/HRNet-Object-Detection"},{"title":"mindspore-lab/mindone","url":"https://github.com/mindspore-lab/mindone"},{"title":"leeyegy/SimDR","url":"https://github.com/leeyegy/SimDR"},{"title":"leeyegy/simcc","url":"https://github.com/leeyegy/simcc"},{"title":"mks0601/PoseFix_RELEASE","url":"https://github.com/mks0601/PoseFix_RELEASE"},{"title":"HRNet/HRNet-Human-Pose-Estimation","url":"https://github.com/HRNet/HRNet-Human-Pose-Estimation"},{"title":"strivebo/image_segmentation_dl","url":"https://github.com/strivebo/image_segmentation_dl"},{"title":"HRNet/HRNet-MaskRCNN-Benchmark","url":"https://github.com/HRNet/HRNet-MaskRCNN-Benchmark"},{"title":"CASIA-IVA-Lab/ISP-reID","url":"https://github.com/CASIA-IVA-Lab/ISP-reID"},{"title":"NVlabs/PAMTRI","url":"https://github.com/NVlabs/PAMTRI"},{"title":"k-miran/hear","url":"https://github.com/k-miran/hear"},{"title":"Vill-Lab/2022-TIP-HCGA","url":"https://github.com/Vill-Lab/2022-TIP-HCGA"},{"title":"d-shivam/Pose-estimation-based-action-recognition-for-help-Situation-Identification","url":"https://github.com/d-shivam/Pose-estimation-based-action-recognition-for-help-Situation-Identification"},{"title":"chuanqichen/deepcoaching","url":"https://github.com/chuanqichen/deepcoaching"},{"title":"Mary-xl/HRnet_Kaggle_iNat2019_FGVC","url":"https://github.com/Mary-xl/HRnet_Kaggle_iNat2019_FGVC"},{"title":"v1viswan/Domain_adaptation_in_HRNet","url":"https://github.com/v1viswan/Domain_adaptation_in_HRNet"},{"title":"ken724049/action-recognition","url":"https://github.com/ken724049/action-recognition"},{"title":"NU-LL/lighttrack-","url":"https://github.com/NU-LL/lighttrack-"},{"title":"thoughtmachines/Human-Pose-Estimation-using-HRNets","url":"https://github.com/thoughtmachines/Human-Pose-Estimation-using-HRNets"},{"title":"ducongju/HRNet","url":"https://github.com/ducongju/HRNet"},{"title":"thomasslloyd/FitSpatial","url":"https://github.com/thomasslloyd/FitSpatial"},{"title":"laowang666888/HRNET","url":"https://github.com/laowang666888/HRNET"},{"title":"baoshengyu/deep-high-resolution-net.pytorch","url":"https://github.com/baoshengyu/deep-high-resolution-net.pytorch"},{"title":"sdll/hrnet-pose-estimation","url":"https://github.com/sdll/hrnet-pose-estimation"},{"title":"gox-ai/hrnet-pose-api","url":"https://github.com/gox-ai/hrnet-pose-api"},{"title":"anshky/HR-NET","url":"https://github.com/anshky/HR-NET"},{"title":"wsjzha/deep-high-resolution-net.pytorch","url":"https://github.com/wsjzha/deep-high-resolution-net.pytorch"},{"title":"visionNoob/hrnet_pytorch","url":"https://github.com/visionNoob/hrnet_pytorch"},{"title":"abhi1kumar/hrnet_pose_single_gpu","url":"https://github.com/abhi1kumar/hrnet_pose_single_gpu"},{"title":"goutern/PoseEstimation","url":"https://github.com/goutern/PoseEstimation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-person-pose-estimation-on-coco","task":"Multi-Person Pose Estimation","dataset_variant":"COCO (Common Objects in Context)","rows":15,"metrics":["AP","Test AP","Validation AP"],"first_row_in_archive_order":{"model":"RSN","paper":"/paper/learning-delicate-local-representations-for","metrics":{"AP":"0.792"},"code_links":[{"title":"open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose"},{"title":"chenyilun95/tf-cpn","url":"https://github.com/chenyilun95/tf-cpn"},{"title":"caiyuanhao1998/RSN","url":"https://github.com/caiyuanhao1998/RSN"},{"title":"HuangJunJie2017/UDP-Pose","url":"https://github.com/HuangJunJie2017/UDP-Pose"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-person-pose-estimation-on-coco-test-dev","task":"Multi-Person Pose Estimation","dataset_variant":"COCO test-dev","rows":15,"metrics":["AP","APL","APM","AP50","AP75","AR","AR50"],"first_row_in_archive_order":{"model":"SCIO (HRNet-48)","paper":"/paper/self-constrained-inference-optimization-on","metrics":{"AP":"79.2","AP50":"93.5","AP75":"85.8","APL":"84.2","APM":"74.1","AR":"81.6"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-coco-visual-4","task":"Visual Question Answering (VQA)","dataset_variant":"COCO Visual Question Answering (VQA) real images 1.0 open ended","rows":14,"metrics":["Percentage correct"],"first_row_in_archive_order":{"model":"MCB 7 att.","paper":"/paper/multimodal-compact-bilinear-pooling-for","metrics":{"Percentage correct":"66.5"},"code_links":[{"title":"Cadene/vqa.pytorch","url":"https://github.com/Cadene/vqa.pytorch"},{"title":"MarcBS/keras","url":"https://github.com/MarcBS/keras"},{"title":"akirafukui/vqa-mcb","url":"https://github.com/akirafukui/vqa-mcb"},{"title":"yikang-li/iqan","url":"https://github.com/yikang-li/iqan"},{"title":"jnhwkim/cbp","url":"https://github.com/jnhwkim/cbp"},{"title":"gabegrand/adversarial-vqa","url":"https://github.com/gabegrand/adversarial-vqa"},{"title":"Adam1679/mutan-article-net","url":"https://github.com/Adam1679/mutan-article-net"},{"title":"vuhoangminh/vqa_medical","url":"https://github.com/vuhoangminh/vqa_medical"},{"title":"arunmallya/simple-vqa","url":"https://github.com/arunmallya/simple-vqa"},{"title":"JoonSeongPark/vqa","url":"https://github.com/JoonSeongPark/vqa"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/pose-estimation-on-coco","task":"Pose Estimation","dataset_variant":"COCO (Common Objects in Context)","rows":10,"metrics":["AP","AR","AP50","AP75","APL","APM"],"first_row_in_archive_order":{"model":"OmniPose (WASPv2)","paper":"/paper/omnipose-a-multi-scale-framework-for-multi","metrics":{"AP":"79.5","AP50":"93.6","AP75":"85.9","APL":"84.6","APM":"76","AR":"81.9"},"code_links":[{"title":"bmartacho/OmniPose","url":"https://github.com/bmartacho/OmniPose"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/single-object-discovery-on-coco-20k","task":"Single-object discovery","dataset_variant":"COCO_20k","rows":10,"metrics":["CorLoc"],"first_row_in_archive_order":{"model":"IMST","paper":"/paper/k-means-for-unsupervised-instance","metrics":{"CorLoc":"72.2"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-coco-visual-1","task":"Visual Question Answering (VQA)","dataset_variant":"COCO Visual Question Answering (VQA) real images 1.0 multiple choice","rows":10,"metrics":["Percentage correct"],"first_row_in_archive_order":{"model":"MCB 7 att.","paper":"/paper/multimodal-compact-bilinear-pooling-for","metrics":{"Percentage correct":"70.1"},"code_links":[{"title":"Cadene/vqa.pytorch","url":"https://github.com/Cadene/vqa.pytorch"},{"title":"MarcBS/keras","url":"https://github.com/MarcBS/keras"},{"title":"akirafukui/vqa-mcb","url":"https://github.com/akirafukui/vqa-mcb"},{"title":"yikang-li/iqan","url":"https://github.com/yikang-li/iqan"},{"title":"jnhwkim/cbp","url":"https://github.com/jnhwkim/cbp"},{"title":"gabegrand/adversarial-vqa","url":"https://github.com/gabegrand/adversarial-vqa"},{"title":"Adam1679/mutan-article-net","url":"https://github.com/Adam1679/mutan-article-net"},{"title":"vuhoangminh/vqa_medical","url":"https://github.com/vuhoangminh/vqa_medical"},{"title":"arunmallya/simple-vqa","url":"https://github.com/arunmallya/simple-vqa"},{"title":"JoonSeongPark/vqa","url":"https://github.com/JoonSeongPark/vqa"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-composed-image-retrieval-zs-cir-on-4","task":"Zero-Shot Composed Image Retrieval (ZS-CIR)","dataset_variant":"COCO (Common Objects in Context)","rows":10,"metrics":["Actions Recall@5"],"first_row_in_archive_order":{"model":"iSEARLE-XL-OTI (CLIP L/14)","paper":"/paper/isearle-improving-textual-inversion-for-zero","metrics":{"Actions Recall@5":"32.55"},"code_links":[{"title":"miccunifi/searle","url":"https://github.com/miccunifi/searle"},{"title":"miccunifi/circo","url":"https://github.com/miccunifi/circo"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-to-text-retrieval-on-coco","task":"Image-to-Text Retrieval","dataset_variant":"COCO (Common Objects in Context)","rows":9,"metrics":["Recall@1","Recall@5","Recall@10"],"first_row_in_archive_order":{"model":"BLIP-2 (ViT-G, fine-tuned)","paper":"/paper/blip-2-bootstrapping-language-image-pre","metrics":{"Recall@1":"85.4","Recall@10":"98.5","Recall@5":"97.0"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"salesforce/lavis","url":"https://github.com/salesforce/lavis"},{"title":"thudm/visualglm-6b","url":"https://github.com/thudm/visualglm-6b"},{"title":"baaivision/eva","url":"https://github.com/baaivision/eva"},{"title":"junshutang/Make-It-3D","url":"https://github.com/junshutang/Make-It-3D"},{"title":"facebookresearch/multimodal","url":"https://github.com/facebookresearch/multimodal"},{"title":"unispac/visual-adversarial-examples-jailbreak-large-language-models","url":"https://github.com/unispac/visual-adversarial-examples-jailbreak-large-language-models"},{"title":"yukw777/videoblip","url":"https://github.com/yukw777/videoblip"},{"title":"alibaba/graphtranslator","url":"https://github.com/alibaba/graphtranslator"},{"title":"gregor-ge/mblip","url":"https://github.com/gregor-ge/mblip"},{"title":"linzhiqiu/clip-flant5","url":"https://github.com/linzhiqiu/clip-flant5"},{"title":"kdr/videorag-mrr2024","url":"https://github.com/kdr/videorag-mrr2024"},{"title":"rabiulcste/vqazero","url":"https://github.com/rabiulcste/vqazero"},{"title":"jiwanchung/vlis","url":"https://github.com/jiwanchung/vlis"},{"title":"yangyucheng000/University","url":"https://github.com/yangyucheng000/University/tree/main/model-2/blip_2"},{"title":"2024-MindSpore-1/Code2","url":"https://github.com/2024-MindSpore-1/Code2/tree/main/model-1/blip_2"},{"title":"albertotestoni/ndq_visual_objects","url":"https://github.com/albertotestoni/ndq_visual_objects"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semantic-segmentation-on-coco-1","task":"Semantic Segmentation","dataset_variant":"COCO (Common Objects in Context)","rows":9,"metrics":["mIoU"],"first_row_in_archive_order":{"model":"HyperSeg","paper":"/paper/hyperseg-towards-universal-visual","metrics":{"mIoU":"77.2"},"code_links":[{"title":"congvvc/HyperSeg","url":"https://github.com/congvvc/HyperSeg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-object-detection-on-ms-coco","task":"Zero-Shot Object Detection","dataset_variant":"MS-COCO","rows":9,"metrics":["mAP","Recall"],"first_row_in_archive_order":{"model":"UniFa","paper":"/paper/unifa-a-unified-feature-hallucination","metrics":{"Recall":"69.10","mAP":"26.00"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/keypoint-detection-on-coco-test-challenge","task":"Keypoint Detection","dataset_variant":"COCO test-challenge","rows":8,"metrics":["AR","ARM","AP","AP50","AP75","APL","AR50","AR75","ARL"],"first_row_in_archive_order":{"model":"4×RSN-50","paper":"/paper/learning-delicate-local-representations-for","metrics":{"AP":"77.1","AP50":"93.3","AP75":"83.6","APL":"82.6","AR":"82.6","AR50":"96.1","AR75":"88.2","ARL":"88.7","ARM":"78.0"},"code_links":[{"title":"open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose"},{"title":"chenyilun95/tf-cpn","url":"https://github.com/chenyilun95/tf-cpn"},{"title":"caiyuanhao1998/RSN","url":"https://github.com/caiyuanhao1998/RSN"},{"title":"HuangJunJie2017/UDP-Pose","url":"https://github.com/HuangJunJie2017/UDP-Pose"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/box-supervised-instance-segmentation-on-coco","task":"Box-supervised Instance Segmentation","dataset_variant":"COCO test-dev","rows":7,"metrics":["mask AP"],"first_row_in_archive_order":{"model":"Box2Mask-T","paper":"/paper/box2mask-box-supervised-instance-segmentation","metrics":{"mask AP":"42.4"},"code_links":[{"title":"LiWentomng/BoxInstSeg","url":"https://github.com/LiWentomng/BoxInstSeg"},{"title":"liwentomng/boxlevelset","url":"https://github.com/liwentomng/boxlevelset"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-level-supervised-instance-segmentation-1","task":"Image-level Supervised Instance Segmentation","dataset_variant":"COCO test-dev","rows":7,"metrics":["AP","AP@50","AP@75"],"first_row_in_archive_order":{"model":"WeakSAM-Mask2Former (with SAM)","paper":"/paper/weaksam-segment-anything-meets-weakly","metrics":{"AP":"25.9","AP@50":"39.9","AP@75":"27.9"},"code_links":[{"title":"hustvl/weaksam","url":"https://github.com/hustvl/weaksam"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-counting-on-coco-count-test","task":"Object Counting","dataset_variant":"COCO count-test","rows":7,"metrics":["m-reIRMSE","m-reIRMSE-nz","mRMSE","mRMSE-nz"],"first_row_in_archive_order":{"model":"ens","paper":"/paper/counting-everyday-objects-in-everyday-scenes","metrics":{"m-reIRMSE":"0.18","m-reIRMSE-nz":"0.81","mRMSE":"0.36","mRMSE-nz":"1.98"},"code_links":[{"title":"prithv1/cvpr2017_counting","url":"https://github.com/prithv1/cvpr2017_counting"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/weakly-supervised-instance-segmentation-on-2","task":"Weakly-supervised instance segmentation","dataset_variant":"COCO test-dev","rows":7,"metrics":["AP","AP@50","AP@75","AP@S","AP@M","AP@L"],"first_row_in_archive_order":{"model":"DiscoBox (ResNeXt-101-DCN-FPN)","paper":"/paper/discobox-weakly-supervised-instance","metrics":{"AP":"37.9","AP@50":"61.4","AP@75":"40.0","AP@L":"53.9","AP@M":"41.1","AP@S":"18.0"},"code_links":[{"title":"NVlabs/DiscoBox","url":"https://github.com/NVlabs/DiscoBox"},{"title":"voidrank/DiscoBox","url":"https://github.com/voidrank/DiscoBox"},{"title":"voidrank/SCOT","url":"https://github.com/voidrank/SCOT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-retrieval-on-coco","task":"Image Retrieval","dataset_variant":"COCO (Common Objects in Context)","rows":6,"metrics":["recall@1","recall@5","Recall@10","QPS"],"first_row_in_archive_order":{"model":"BLIP-2 ViT-G (fine-tuned)","paper":"/paper/blip-2-bootstrapping-language-image-pre","metrics":{"Recall@10":"92.6","recall@1":"68.3","recall@5":"87.7"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"salesforce/lavis","url":"https://github.com/salesforce/lavis"},{"title":"thudm/visualglm-6b","url":"https://github.com/thudm/visualglm-6b"},{"title":"baaivision/eva","url":"https://github.com/baaivision/eva"},{"title":"junshutang/Make-It-3D","url":"https://github.com/junshutang/Make-It-3D"},{"title":"facebookresearch/multimodal","url":"https://github.com/facebookresearch/multimodal"},{"title":"unispac/visual-adversarial-examples-jailbreak-large-language-models","url":"https://github.com/unispac/visual-adversarial-examples-jailbreak-large-language-models"},{"title":"yukw777/videoblip","url":"https://github.com/yukw777/videoblip"},{"title":"alibaba/graphtranslator","url":"https://github.com/alibaba/graphtranslator"},{"title":"gregor-ge/mblip","url":"https://github.com/gregor-ge/mblip"},{"title":"linzhiqiu/clip-flant5","url":"https://github.com/linzhiqiu/clip-flant5"},{"title":"kdr/videorag-mrr2024","url":"https://github.com/kdr/videorag-mrr2024"},{"title":"rabiulcste/vqazero","url":"https://github.com/rabiulcste/vqazero"},{"title":"jiwanchung/vlis","url":"https://github.com/jiwanchung/vlis"},{"title":"yangyucheng000/University","url":"https://github.com/yangyucheng000/University/tree/main/model-2/blip_2"},{"title":"2024-MindSpore-1/Code2","url":"https://github.com/2024-MindSpore-1/Code2/tree/main/model-1/blip_2"},{"title":"albertotestoni/ndq_visual_objects","url":"https://github.com/albertotestoni/ndq_visual_objects"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-semantic-segmentation-on-coco-1","task":"Unsupervised Semantic Segmentation","dataset_variant":"COCO-Stuff-3","rows":6,"metrics":["Pixel Accuracy"],"first_row_in_archive_order":{"model":"SAN","paper":"/paper/rethinking-alignment-and-uniformity-in","metrics":{"Pixel Accuracy":"80.3"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/layout-to-image-generation-on-coco-stuff-4","task":"Layout-to-Image Generation","dataset_variant":"COCO-Stuff 256x256","rows":5,"metrics":["FID","Inception Score","LPIPS"],"first_row_in_archive_order":{"model":"LayoutDiffusion (25steps)","paper":"/paper/layoutdiffusion-controllable-diffusion-model","metrics":{"FID":"31.68"},"code_links":[{"title":"zgctroy/layoutdiffusion","url":"https://github.com/zgctroy/layoutdiffusion"},{"title":"dcdcvgroup/layout-diffusion-mindspore","url":"https://github.com/dcdcvgroup/layout-diffusion-mindspore"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/weakly-supervised-object-detection-on-coco","task":"Weakly Supervised Object Detection","dataset_variant":"COCO (Common Objects in Context)","rows":5,"metrics":["MAP"],"first_row_in_archive_order":{"model":"MSLPD","paper":"/paper/few-example-object-detection-with-model","metrics":{"MAP":"56.6"},"code_links":[{"title":"D-X-Y/DXY-Projects","url":"https://github.com/D-X-Y/DXY-Projects/tree/master/MSPLD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/knowledge-distillation-on-coco","task":"Knowledge Distillation","dataset_variant":"COCO (Common Objects in Context)","rows":4,"metrics":[" box AP","mask AP","mAP"],"first_row_in_archive_order":{"model":"ADLIK-Faster (T: Faster R-CNN vit-base S: Faster R-CNN deit-small)","paper":"/paper/focal-and-global-knowledge-distillation-for","metrics":{" box AP":"47.6"},"code_links":[{"title":"yzd-v/FGD","url":"https://github.com/yzd-v/FGD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-person-pose-estimation-on-coco-minival","task":"Multi-Person Pose Estimation","dataset_variant":"COCO minival","rows":4,"metrics":["AP"],"first_row_in_archive_order":{"model":"HRNet-W48plus","paper":"/paper/how-to-train-your-robust-human-pose-estimator","metrics":{"AP":"79.1"},"code_links":[{"title":"open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose"},{"title":"HuangJunJie2017/UDP-Pose","url":"https://github.com/HuangJunJie2017/UDP-Pose"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/one-shot-object-detection-on-coco","task":"One-Shot Object Detection","dataset_variant":"COCO (Common Objects in Context)","rows":4,"metrics":["AP 0.5"],"first_row_in_archive_order":{"model":"OWL-ViT (R50+H/32)","paper":"/paper/simple-open-vocabulary-object-detection-with","metrics":{"AP 0.5":"41.8"},"code_links":[{"title":"google-research/scenic","url":"https://github.com/google-research/scenic/tree/main/scenic/projects/owl_vit"},{"title":"yangyucheng000/University","url":"https://github.com/yangyucheng000/University/tree/main/model-1/owlvit"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/question-generation-on-coco-visual-question","task":"Question Generation","dataset_variant":"COCO Visual Question Answering (VQA) real images 1.0 open ended","rows":4,"metrics":["BLEU-1"],"first_row_in_archive_order":{"model":"MDN","paper":"/paper/multimodal-differential-network-for-visual","metrics":{"BLEU-1":"65.1"},"code_links":[{"title":"badripatro/MDN-VQG","url":"https://github.com/badripatro/MDN-VQG"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-coco-visual","task":"Visual Question Answering (VQA)","dataset_variant":"COCO Visual Question Answering (VQA) real images 2.0 open ended","rows":4,"metrics":["Percentage correct"],"first_row_in_archive_order":{"model":"HDU-USYD-UNCC","paper":"/paper/vqa-visual-question-answering","metrics":{"Percentage correct":"68.16"},"code_links":[{"title":"ramprs/grad-cam","url":"https://github.com/ramprs/grad-cam"},{"title":"abhshkdz/neural-vqa-attention","url":"https://github.com/abhshkdz/neural-vqa-attention"},{"title":"tbmoon/basic_vqa","url":"https://github.com/tbmoon/basic_vqa"},{"title":"Shivanshu-Gupta/Visual-Question-Answering","url":"https://github.com/Shivanshu-Gupta/Visual-Question-Answering"},{"title":"vipulgupta1011/swapmix","url":"https://github.com/vipulgupta1011/swapmix"},{"title":"ntusteeian/VQA_CNN-LSTM","url":"https://github.com/ntusteeian/VQA_CNN-LSTM"},{"title":"SatyamGaba/visual_question_answering","url":"https://github.com/SatyamGaba/visual_question_answering"},{"title":"SatyamGaba/vqa","url":"https://github.com/SatyamGaba/vqa"},{"title":"Shivmohith/Visual-Assistance-for-the-Blind","url":"https://github.com/Shivmohith/Visual-Assistance-for-the-Blind"},{"title":"mokhalid-dev/Attention-based-VQA-model","url":"https://github.com/mokhalid-dev/Attention-based-VQA-model"},{"title":"SDaydreamer/VisualQA_Project","url":"https://github.com/SDaydreamer/VisualQA_Project"},{"title":"abhijit-buet/VizWiz-Visual-Question-Answering-2021","url":"https://github.com/abhijit-buet/VizWiz-Visual-Question-Answering-2021"},{"title":"mkhalil1998/EC601_Group_Project","url":"https://github.com/mkhalil1998/EC601_Group_Project"},{"title":"abhijit-buet/VizWiz-Visua-Question-Answering-2021","url":"https://github.com/abhijit-buet/VizWiz-Visua-Question-Answering-2021"},{"title":"chirag26495/DAN_VQA","url":"https://github.com/chirag26495/DAN_VQA"},{"title":"yanxinyan1/yxy","url":"https://github.com/yanxinyan1/yxy"},{"title":"moh833/VQA","url":"https://github.com/moh833/VQA"},{"title":"mishajw/vocab_pie","url":"https://github.com/mishajw/vocab_pie"},{"title":"SuchismitaSahu1993/VQA-System","url":"https://github.com/SuchismitaSahu1993/VQA-System"},{"title":"ruxuan666/VQA_program","url":"https://github.com/ruxuan666/VQA_program"},{"title":"luomancs/alternative_answer_set","url":"https://github.com/luomancs/alternative_answer_set"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-coco-visual-2","task":"Visual Question Answering (VQA)","dataset_variant":"COCO Visual Question Answering (VQA) abstract images 1.0 open ended","rows":4,"metrics":["Percentage correct"],"first_row_in_archive_order":{"model":"Graph VQA","paper":"/paper/graph-structured-representations-for-visual","metrics":{"Percentage correct":"70.42"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-coco-visual-3","task":"Visual Question Answering (VQA)","dataset_variant":"COCO Visual Question Answering (VQA) abstract 1.0 multiple choice","rows":4,"metrics":["Percentage correct"],"first_row_in_archive_order":{"model":"Graph VQA","paper":"/paper/graph-structured-representations-for-visual","metrics":{"Percentage correct":"74.37"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/weakly-supervised-object-detection-on-coco-2","task":"Weakly Supervised Object Detection","dataset_variant":"COCO test-dev","rows":4,"metrics":["AP50"],"first_row_in_archive_order":{"model":"wetectron(single-model, VGG16)","paper":"/paper/instance-aware-context-focused-and-memory","metrics":{"AP50":"24.8"},"code_links":[{"title":"NVlabs/wetectron","url":"https://github.com/NVlabs/wetectron"},{"title":"ppengtang/pcl.pytorch","url":"https://github.com/ppengtang/pcl.pytorch"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-detection-on-coco-1","task":"Object Detection","dataset_variant":"COCO (Common Objects in Context)","rows":3,"metrics":["box AP","GFlops"],"first_row_in_archive_order":{"model":"MOAT-3 22K+1K","paper":"/paper/moat-alternating-mobile-convolution-and","metrics":{"box AP":"59.2"},"code_links":[{"title":"google-research/deeplab2","url":"https://github.com/google-research/deeplab2"},{"title":"RooKichenn/pytorch-MOAT","url":"https://github.com/RooKichenn/pytorch-MOAT"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/conditional-image-generation-on-coco-animals","task":"Conditional Image Generation","dataset_variant":"COCO-Animals","rows":2,"metrics":["FID","IS"],"first_row_in_archive_order":{"model":"U-Net GAN","paper":"/paper/a-u-net-based-discriminator-for-generative","metrics":{"FID":"13.73","IS":"12.29"},"code_links":[{"title":"boschresearch/unetgan","url":"https://github.com/boschresearch/unetgan"},{"title":"xingchenzhao/Generating-Human-Skeletons-with-Mutual-Actions-WGAN-Pytorch","url":"https://github.com/xingchenzhao/Generating-Human-Skeletons-with-Mutual-Actions-WGAN-Pytorch"},{"title":"xingchenzhao/deep-learning-team-project","url":"https://github.com/xingchenzhao/deep-learning-team-project"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/cross-modal-retrieval-on-mscoco-1k","task":"Cross-Modal Retrieval","dataset_variant":"MSCOCO-1k","rows":2,"metrics":["Image-to-text R@1","Text-to-image R@1"],"first_row_in_archive_order":{"model":"NAPReg","paper":"/paper/napreg-nouns-as-proxies-regularization-for","metrics":{"Image-to-text R@1":"81.9","Text-to-image R@1":"66.9"},"code_links":[{"title":"bhavinjawade/NAPReq","url":"https://github.com/bhavinjawade/NAPReq"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-world-object-detection-on-coco-2017","task":"Open World Object Detection","dataset_variant":"COCO 2017 (Outdoor, Accessories, Appliance, Truck)","rows":2,"metrics":["Unknown Recall","MAP","WI","A-OSE"],"first_row_in_archive_order":{"model":"ORE (MDef-DETR)","paper":"/paper/multi-modal-transformers-excel-at-class","metrics":{"A-OSE":"5212","MAP":"46.19","Unknown Recall":"49.54","WI":"0.0251"},"code_links":[{"title":"mmaaz60/mvits_for_class_agnostic_od","url":"https://github.com/mmaaz60/mvits_for_class_agnostic_od"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-world-object-detection-on-coco-2017-1","task":"Open World Object Detection","dataset_variant":"COCO 2017 (Sports, Food)","rows":2,"metrics":["Unknown Recall","MAP","WI","A-OSE"],"first_row_in_archive_order":{"model":"ORE (MDef-DETR)","paper":"/paper/multi-modal-transformers-excel-at-class","metrics":{"A-OSE":"4117","MAP":"36.75","Unknown Recall":"50.89","WI":"0.0179"},"code_links":[{"title":"mmaaz60/mvits_for_class_agnostic_od","url":"https://github.com/mmaaz60/mvits_for_class_agnostic_od"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-world-object-detection-on-coco-2017-2","task":"Open World Object Detection","dataset_variant":"COCO 2017 (Electronic, Indoor, Kitchen, Furniture)","rows":2,"metrics":["MAP"],"first_row_in_archive_order":{"model":"ORE (MDef-DETR)","paper":"/paper/multi-modal-transformers-excel-at-class","metrics":{"MAP":"31.66"},"code_links":[{"title":"mmaaz60/mvits_for_class_agnostic_od","url":"https://github.com/mmaaz60/mvits_for_class_agnostic_od"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/panoptic-segmentation-on-coco-panoptic","task":"Panoptic Segmentation","dataset_variant":"COCO panoptic","rows":2,"metrics":["PQ","PQst","PQth"],"first_row_in_archive_order":{"model":"VAN-B6*","paper":"/paper/visual-attention-network","metrics":{"PQ":"58.2"},"code_links":[{"title":"huggingface/transformers","url":"https://github.com/huggingface/transformers"},{"title":"facebookresearch/xformers","url":"https://github.com/facebookresearch/xformers"},{"title":"PaddlePaddle/PaddleClas","url":"https://github.com/PaddlePaddle/PaddleClas"},{"title":"open-mmlab/mmclassification","url":"https://github.com/open-mmlab/mmclassification"},{"title":"lucasjinreal/yolov7_d2","url":"https://github.com/lucasjinreal/yolov7_d2"},{"title":"MenghaoGuo/Awesome-Vision-Attentions","url":"https://github.com/MenghaoGuo/Awesome-Vision-Attentions"},{"title":"chengtan9907/simvpv2","url":"https://github.com/chengtan9907/simvpv2"},{"title":"sithu31296/semantic-segmentation","url":"https://github.com/sithu31296/semantic-segmentation"},{"title":"Visual-Attention-Network/VAN-Classification","url":"https://github.com/Visual-Attention-Network/VAN-Classification"},{"title":"Westlake-AI/openmixup","url":"https://github.com/Westlake-AI/openmixup"},{"title":"Visual-Attention-Network/VAN-Segmentation","url":"https://github.com/Visual-Attention-Network/VAN-Segmentation"},{"title":"DarshanDeshpande/jax-models","url":"https://github.com/DarshanDeshpande/jax-models"},{"title":"Jittor-Image-Models/Jittor-Image-Models","url":"https://github.com/Jittor-Image-Models/Jittor-Image-Models"},{"title":"Visual-Attention-Network/VAN-Jittor","url":"https://github.com/Visual-Attention-Network/VAN-Jittor"},{"title":"birder/birder","url":"https://gitlab.com/birder/birder"},{"title":"shkarupa-alex/tfvan","url":"https://github.com/shkarupa-alex/tfvan"},{"title":"pwc-1/Paper-9","url":"https://github.com/pwc-1/Paper-9/tree/main/5/van"},{"title":"Asthestarsfalll/VAN-MegEngine","url":"https://github.com/Asthestarsfalll/VAN-MegEngine"},{"title":"flytocc/PaddleClas","url":"https://github.com/flytocc/PaddleClas"},{"title":"EMalagoli92/VAN-Classification-TensorFlow","url":"https://github.com/EMalagoli92/VAN-Classification-TensorFlow"},{"title":"MindCode-4/code-1","url":"https://github.com/MindCode-4/code-1/tree/main/van"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/robust-object-detection-on-coco","task":"Robust Object Detection","dataset_variant":"COCO (Common Objects in Context)","rows":2,"metrics":["mPC [AP]","rPC [%]"],"first_row_in_archive_order":{"model":"Faster R-CNN with Stylized Training Data","paper":"/paper/benchmarking-robustness-in-object-detection","metrics":{"mPC [AP]":"20.4","rPC [%]":"58.9"},"code_links":[{"title":"bethgelab/imagecorruptions","url":"https://github.com/bethgelab/imagecorruptions"},{"title":"bethgelab/robust-detection-benchmark","url":"https://github.com/bethgelab/robust-detection-benchmark"},{"title":"bethgelab/stylize-datasets","url":"https://github.com/bethgelab/stylize-datasets"},{"title":"bethgelab/mmdetection","url":"https://github.com/bethgelab/mmdetection"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-to-image-generation-on-ms-coco","task":"Text-to-Image Generation","dataset_variant":"MS-COCO","rows":2,"metrics":["Inception score","FID","SOA-C"],"first_row_in_archive_order":{"model":"AttnGAN","paper":"/paper/attngan-fine-grained-text-to-image-generation","metrics":{"FID":"35.49","Inception score":"25.89","SOA-C":"25.88"},"code_links":[{"title":"taoxugit/AttnGAN","url":"https://github.com/taoxugit/AttnGAN"},{"title":"davidstap/AttnGAN","url":"https://github.com/davidstap/AttnGAN"},{"title":"sidward14/Style-AttnGAN","url":"https://github.com/sidward14/Style-AttnGAN"},{"title":"huiyegit/T2I_CL","url":"https://github.com/huiyegit/T2I_CL"},{"title":"Shuvrajit9904/PairedCycleGAN-tf","url":"https://github.com/Shuvrajit9904/PairedCycleGAN-tf"},{"title":"taki0112/AttnGAN-Tensorflow","url":"https://github.com/taki0112/AttnGAN-Tensorflow"},{"title":"komiya-m/MirrorGAN","url":"https://github.com/komiya-m/MirrorGAN"},{"title":"bprabhakar/text-to-image","url":"https://github.com/bprabhakar/text-to-image"},{"title":"oxygenlu/ratlip","url":"https://github.com/oxygenlu/ratlip"},{"title":"pioneerAlpha/BanglaText2ImageGeneration","url":"https://github.com/pioneerAlpha/BanglaText2ImageGeneration"},{"title":"aleksey-egorov/attngan","url":"https://github.com/aleksey-egorov/attngan"},{"title":"priscillalui/StackGAN-Stories","url":"https://github.com/priscillalui/StackGAN-Stories"},{"title":"ChihchengHsieh/AttnGAN_Implementation","url":"https://github.com/ChihchengHsieh/AttnGAN_Implementation"},{"title":"roxanasoto/AttGanESRGAN","url":"https://github.com/roxanasoto/AttGanESRGAN"},{"title":"Maymaher/StackGANv2","url":"https://github.com/Maymaher/StackGANv2"},{"title":"Vigneshthanga/stackGAN-v2","url":"https://github.com/Vigneshthanga/stackGAN-v2"},{"title":"ucsd-ml-arts/ml-art-final-jeffrey","url":"https://github.com/ucsd-ml-arts/ml-art-final-jeffrey"},{"title":"rightlit/cycle-image-gan-rev","url":"https://github.com/rightlit/cycle-image-gan-rev"},{"title":"alexmotogna/generatorapi","url":"https://github.com/alexmotogna/generatorapi"},{"title":"alexmotogna/attngan","url":"https://github.com/alexmotogna/attngan"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/active-object-detection-on-coco","task":"Active Object Detection","dataset_variant":"COCO (Common Objects in Context)","rows":1,"metrics":["AP"],"first_row_in_archive_order":{"model":"RetinaNet","paper":"/paper/multiple-instance-active-learning-for-object","metrics":{"AP":"(7.3, 13.8, 16.9, 19.1, 20.8) on 2% ~ 10%"},"code_links":[{"title":"yuantn/MI-AOD","url":"https://github.com/yuantn/MI-AOD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/activeness-detection-on-coco-test-dev","task":"Activeness Detection","dataset_variant":"COCO test-dev","rows":1,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"Lightweight OpenPose","paper":"/paper/activenet-a-computer-vision-based-approach-to","metrics":{"Accuracy (%)":"76.67"},"code_links":[{"title":"aaditagarwal/ActiveNet","url":"https://github.com/aaditagarwal/ActiveNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/few-shot-object-detection-on-coco-2017","task":"Few-Shot Object Detection","dataset_variant":"COCO 2017","rows":1,"metrics":["AP"],"first_row_in_archive_order":{"model":"DETReg (ours)","paper":"/paper/detreg-unsupervised-pretraining-with-region","metrics":{"AP":"30"},"code_links":[{"title":"amirbar/detreg","url":"https://github.com/amirbar/detreg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/homography-estimation-on-coco-2014","task":"Homography Estimation","dataset_variant":"COCO 2014","rows":1,"metrics":["MACE"],"first_row_in_archive_order":{"model":"PFNet","paper":"/paper/rethinking-planar-homography-estimation-using","metrics":{"MACE":"0.92"},"code_links":[{"title":"ruizengalways/PFNet","url":"https://github.com/ruizengalways/PFNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-captioning-on-ms-coco","task":"Image Captioning","dataset_variant":"MS-COCO","rows":1,"metrics":["BLEU-1","BLEU-4","CIDEr","METEOR","SPICE","Test ROGUE-L"],"first_row_in_archive_order":{"model":"NeuSyRE","paper":"/paper/neusyre-neuro-symbolic-visual-understanding","metrics":{"BLEU-1":"79.1","BLEU-4":"37.6","CIDEr":"131.4","METEOR":"28.5","SPICE":"23.8","Test ROGUE-L":"57.7"},"code_links":[{"title":"jaleedkhan/neusire","url":"https://github.com/jaleedkhan/neusire"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/instance-segmentation-on-coco-minval","task":"Instance Segmentation","dataset_variant":"coco minval","rows":1,"metrics":["APL"],"first_row_in_archive_order":{"model":"R3-CNN (ResNet-50-FPN, GC-Net)","paper":"/paper/recursively-refined-r-cnn-instance","metrics":{"APL":"56"},"code_links":[{"title":"IMPLabUniPr/mmdetection","url":"https://github.com/IMPLabUniPr/mmdetection"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/interactive-segmentation-on-coco","task":"Interactive Segmentation","dataset_variant":"COCO (Common Objects in Context)","rows":1,"metrics":["Instance Average IoU"],"first_row_in_archive_order":{"model":"IOG","paper":"/paper/interactive-object-segmentation-with-inside","metrics":{"Instance Average IoU":"85.2"},"code_links":[{"title":"shiyinzhang/Inside-Outside-Guidance","url":"https://github.com/shiyinzhang/Inside-Outside-Guidance"},{"title":"shiyinzhang/Pixel-ImageNet","url":"https://github.com/shiyinzhang/Pixel-ImageNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/interactive-segmentation-on-coco-minival","task":"Interactive Segmentation","dataset_variant":"COCO minival","rows":1,"metrics":["NoC@85","NoC@90"],"first_row_in_archive_order":{"model":"ViT-B+MST+CL","paper":"/paper/mst-adaptive-multi-scale-tokens-guided","metrics":{"NoC@85":"2.08","NoC@90":"2.85"},"code_links":[{"title":"hahamyt/mst","url":"https://github.com/hahamyt/mst"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-label-learning-on-coco-2014","task":"Multi-Label Learning","dataset_variant":"COCO 2014","rows":1,"metrics":["CF1","CP","CR","OF1","OP","OR","mAP"],"first_row_in_archive_order":{"model":"SADCL","paper":"/paper/semantic-aware-dual-contrastive-learning-for","metrics":{"CF1":"79.8","CP":"84.6","CR":"76","OF1":"82.1","OP":"86","OR":"78.5","mAP":"85.6"},"code_links":[{"title":"yu-gi-oh-leilei/sadcl","url":"https://github.com/yu-gi-oh-leilei/sadcl"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-object-discovery-on-coco-20k","task":"Multi-object discovery","dataset_variant":"COCO_20k","rows":1,"metrics":["Detection Rate"],"first_row_in_archive_order":{"model":"Large-scale rOSD","paper":"/paper/toward-unsupervised-multi-object-discovery-in","metrics":{"Detection Rate":"12.0"},"code_links":[{"title":"huyvvo/rOSD","url":"https://github.com/huyvvo/rOSD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-detection-on-coco-5","task":"Object Detection","dataset_variant":"COCO+","rows":1,"metrics":["mAR (COCO+ XS)"],"first_row_in_archive_order":{"model":"RepPoints + Self-adaptation","paper":"/paper/slender-object-detection-diagnoses-and","metrics":{"mAR (COCO+ XS)":"28.4"},"code_links":[{"title":"wanzysky/SlenderObjDet","url":"https://github.com/wanzysky/SlenderObjDet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/object-proposal-generation-on-coco","task":"Object Proposal Generation","dataset_variant":"COCO (Common Objects in Context)","rows":1,"metrics":["Average Recall"],"first_row_in_archive_order":{"model":"MDef-DETR (Off-the-shelf evaluation)","paper":"/paper/multi-modal-transformers-excel-at-class","metrics":{"Average Recall":"0.6503"},"code_links":[{"title":"mmaaz60/mvits_for_class_agnostic_od","url":"https://github.com/mmaaz60/mvits_for_class_agnostic_od"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/one-shot-instance-segmentation-on-coco","task":"One-Shot Instance Segmentation","dataset_variant":"COCO (Common Objects in Context)","rows":1,"metrics":["AP 0.5"],"first_row_in_archive_order":{"model":"Siamese Mask R-CNN","paper":"/paper/one-shot-instance-segmentation","metrics":{"AP 0.5":"14.5"},"code_links":[{"title":"bethgelab/siamese-mask-rcnn","url":"https://github.com/bethgelab/siamese-mask-rcnn"},{"title":"michaelisc/cluttered-omniglot","url":"https://github.com/michaelisc/cluttered-omniglot"},{"title":"ducminhkhoi/FAPIS","url":"https://github.com/ducminhkhoi/FAPIS"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/point-supervised-instance-segmentation-on-1","task":"Point-Supervised Instance Segmentation","dataset_variant":"COCO test-dev","rows":1,"metrics":["AP","AP@50","AP@75"],"first_row_in_archive_order":{"model":"BESTIE (proposal-free)","paper":"/paper/beyond-semantic-to-instance-segmentation","metrics":{"AP":"17.8","AP@50":"34.1","AP@75":"16.7"},"code_links":[{"title":"clovaai/BESTIE","url":"https://github.com/clovaai/BESTIE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/pose-estimation-on-coco-minival","task":"Pose Estimation","dataset_variant":"COCO minival","rows":1,"metrics":["AP"],"first_row_in_archive_order":{"model":"MSPN","paper":"/paper/rethinking-on-multi-stage-networks-for-human","metrics":{"AP":"75.9"},"code_links":[{"title":"open-mmlab/mmpose","url":"https://github.com/open-mmlab/mmpose"},{"title":"chenyilun95/tf-cpn","url":"https://github.com/chenyilun95/tf-cpn"},{"title":"megvii-detection/MSPN","url":"https://github.com/megvii-detection/MSPN"},{"title":"fenglinglwb/MSPN","url":"https://github.com/fenglinglwb/MSPN"},{"title":"xiuyu0000/papers_with_examples","url":"https://github.com/xiuyu0000/papers_with_examples/tree/main/mspn"},{"title":"hyperionfalling/lightpose","url":"https://github.com/hyperionfalling/lightpose"},{"title":"yangyucheng000/MSPN","url":"https://github.com/yangyucheng000/MSPN"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/pose-estimation-on-ms-coco","task":"Pose Estimation","dataset_variant":"MS-COCO","rows":1,"metrics":["AP"],"first_row_in_archive_order":{"model":"UniHCP (finetune)","paper":"/paper/unihcp-a-unified-model-for-human-centric","metrics":{"AP":"76.5"},"code_links":[{"title":"opengvlab/unihcp","url":"https://github.com/opengvlab/unihcp"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/quantization-on-coco","task":"Quantization","dataset_variant":"COCO (Common Objects in Context)","rows":1,"metrics":["MAP"],"first_row_in_archive_order":{"model":"SSD ResNet50 V1 FPN 640x640","paper":"/paper/hptq-hardware-friendly-post-training","metrics":{"MAP":"34.3"},"code_links":[{"title":"sony/model_optimization","url":"https://github.com/sony/model_optimization"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/question-answering-on-coco-visual-question","task":"Question Answering","dataset_variant":"COCO Visual Question Answering (VQA) real images 1.0 open ended","rows":1,"metrics":["Test"],"first_row_in_archive_order":{"model":"MaMMUT (2B)","paper":"/paper/mammut-a-simple-architecture-for-joint","metrics":{"Test":"80.8"},"code_links":[{"title":"lucidrains/mammut-pytorch","url":"https://github.com/lucidrains/mammut-pytorch"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/real-time-instance-segmentation-on-mscoco-1k","task":"Real-time Instance Segmentation","dataset_variant":"MSCOCO-1k","rows":1,"metrics":["APM"],"first_row_in_archive_order":{"model":"RTMDet-Ins-x","paper":"/paper/rtmdet-an-empirical-study-of-designing-real","metrics":{"APM":"49.0"},"code_links":[{"title":"open-mmlab/mmdetection","url":"https://github.com/open-mmlab/mmdetection/tree/3.x/configs/rtmdet"},{"title":"open-mmlab/mmyolo","url":"https://github.com/open-mmlab/mmyolo"},{"title":"open-mmlab/mmrotate","url":"https://github.com/open-mmlab/mmrotate"},{"title":"open-edge-platform/training_extensions","url":"https://github.com/open-edge-platform/training_extensions"},{"title":"PaddlePaddle/PaddleYOLO","url":"https://github.com/PaddlePaddle/PaddleYOLO"},{"title":"open-edge-platform/geti","url":"https://github.com/open-edge-platform/geti"},{"title":"yxb-nku/strip-r-cnn","url":"https://github.com/yxb-nku/strip-r-cnn"},{"title":"HVision-NKU/Strip-R-CNN","url":"https://github.com/HVision-NKU/Strip-R-CNN"},{"title":"fiveai/MoCaE","url":"https://github.com/fiveai/MoCaE"},{"title":"yuyi1005/point2rbox-mmrotate","url":"https://github.com/yuyi1005/point2rbox-mmrotate"},{"title":"cszzshi/SimD","url":"https://github.com/cszzshi/SimD"},{"title":"CycloneBoy/PPDetectionPytorch","url":"https://github.com/CycloneBoy/PPDetectionPytorch"},{"title":"V3Det/mmdetection-V3Det","url":"https://github.com/V3Det/mmdetection-V3Det"},{"title":"RangiLyu/mmdetection_test","url":"https://github.com/RangiLyu/mmdetection_test"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/scene-graph-generation-on-ms-coco","task":"Scene Graph Generation","dataset_variant":"MS-COCO","rows":1,"metrics":["R@100","R@20","R@50","mR@100","mR@20","mR@50"],"first_row_in_archive_order":{"model":"NeuSyRE","paper":"/paper/neusyre-neuro-symbolic-visual-understanding","metrics":{"R@100":"38.5","R@20":"27.9","R@50":"36.3","mR@100":"12.8","mR@20":"9.2","mR@50":"11.6"},"code_links":[{"title":"jaleedkhan/neusire","url":"https://github.com/jaleedkhan/neusire"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/semi-supervised-learning-for-image-captioning","task":"Semi Supervised Learning for Image Captioning","dataset_variant":"COCO (Common Objects in Context)","rows":1,"metrics":["CIDEr"],"first_row_in_archive_order":{"model":"Perturb, Predict & Paraphrase","paper":"/paper/perturb-predict-paraphrase-semi-supervised","metrics":{"CIDEr":"84.5"},"code_links":[{"title":"csalt-research/perturb-predict-paraphrase","url":"https://github.com/csalt-research/perturb-predict-paraphrase"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-object-localization-on-coco-20k","task":"Unsupervised Object Localization","dataset_variant":"COCO_20k","rows":1,"metrics":["CorLoc"],"first_row_in_archive_order":{"model":"DeepCut","paper":"/paper/deepcut-unsupervised-segmentation-using-graph","metrics":{"CorLoc":"61.6"},"code_links":[{"title":"sampl-weizmann/deepcut","url":"https://github.com/sampl-weizmann/deepcut"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/unsupervised-semantic-segmentation-with-5","task":"Unsupervised Semantic Segmentation with Language-image Pre-training","dataset_variant":"COCO (Common Objects in Context)","rows":1,"metrics":["Mean IoU (val)"],"first_row_in_archive_order":{"model":"CLIPpy ViT-B","paper":"/paper/perceptual-grouping-in-vision-language-models","metrics":{"Mean IoU (val)":"25.5"},"code_links":[{"title":"kahnchana/clippy","url":"https://github.com/kahnchana/clippy"},{"title":"jongwoopark7978/LVNet","url":"https://github.com/jongwoopark7978/LVNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-on-coco-visual-5","task":"Visual Question Answering","dataset_variant":"COCO Visual Question Answering (VQA) real images 2.0 open ended","rows":1,"metrics":["Percentage correct"],"first_row_in_archive_order":{"model":"MaMMUT (2B)","paper":"/paper/mammut-a-simple-architecture-for-joint","metrics":{"Percentage correct":"80.7"},"code_links":[{"title":"lucidrains/mammut-pytorch","url":"https://github.com/lucidrains/mammut-pytorch"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/no-time-to-train-training-free-reference","title":"No time to train! Training-Free Reference-Based Instance Segmentation","date":"2025-07-03","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/the-missing-point-in-vision-transformers-for","title":"The Missing Point in Vision Transformers for Universal Image Segmentation","date":"2025-05-26","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/posebh-prototypical-multi-dataset-training","title":"PoseBH: Prototypical Multi-Dataset Training Beyond Human Pose Estimation","date":"2025-05-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/decoupling-classifier-for-boosting-few-shot-1","title":"Decoupling Classifier for Boosting Few-shot Object Detection and Instance Segmentation","date":"2025-05-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/perception-encoder-the-best-visual-embeddings","title":"Perception Encoder: The best visual embeddings are not at the output of the network","date":"2025-04-17","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":12,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/your-vit-is-secretly-an-image-segmentation-1","title":"Your ViT is Secretly an Image Segmentation Model","date":"2025-03-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unifa-a-unified-feature-hallucination","title":"UniFa: A unified feature hallucination framework for any-shot object detection","date":"2025-03-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/yolov12-a-breakdown-of-the-key-architectural","title":"YOLOv12: A Breakdown of the Key Architectural Features","date":"2025-02-20","rows_on_this_dataset":5,"code_links":0,"syntology":null},{"paper":"/paper/cp-detr-concept-prompt-guide-detr-toward","title":"CP-DETR: Concept Prompt Guide DETR Toward Stronger Universal Object Detection","date":"2024-12-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/deim-detr-with-improved-matching-for-fast","title":"DEIM: DETR with Improved Matching for Fast Convergence","date":"2024-12-05","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":1,"samples_unverified":12,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cosmos-cross-modality-self-distillation-for","title":"COSMOS: Cross-Modality Self-Distillation for Vision Language Pre-training","date":"2024-12-02","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hyperseg-towards-universal-visual","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","date":"2024-11-26","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":7,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/yolov11-an-overview-of-the-key-architectural","title":"YOLOv11: An Overview of the Key Architectural Enhancements","date":"2024-10-23","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/d-fine-redefine-regression-task-in-detrs-as","title":"D-FINE: Redefine Regression Task in DETRs as Fine-grained Distribution Refinement","date":"2024-10-17","rows_on_this_dataset":7,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":10,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/debiformer-vision-transformer-with-deformable","title":"DeBiFormer: Vision Transformer with Deformable Agent Bi-level Routing Attention","date":"2024-10-11","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/data-extrapolation-for-text-to-image","title":"Data Extrapolation for Text-to-image Generation on Small Datasets","date":"2024-10-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/stochastic-subsampling-with-average-pooling","title":"Stochastic Subsampling With Average Pooling","date":"2024-09-25","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/sapiens-foundation-for-human-vision-models","title":"Sapiens: Foundation for Human Vision Models","date":"2024-08-22","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/relation-detr-exploring-explicit-position","title":"Relation DETR: Exploring Explicit Position Relation Prior for Object Detection","date":"2024-07-16","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-branch-auxiliary-fusion-yolo-with-re","title":"Multi-Branch Auxiliary Fusion YOLO with Re-parameterization Heterogeneous Convolutional for accurate object detection","date":"2024-07-05","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/leyolo-new-scalable-and-efficient-cnn","title":"LeYOLO, New Scalable and Efficient CNN Architecture for Object Detection","date":"2024-06-20","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/parameter-inverted-image-pyramid-networks","title":"Parameter-Inverted Image Pyramid Networks","date":"2024-06-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/long-and-short-guidance-in-score-identity","title":"Long and Short Guidance in Score identity Distillation for One-Step Text-to-Image Generation","date":"2024-06-03","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":18,"samples_unverified":6,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/yolov10-real-time-end-to-end-object-detection","title":"YOLOv10: Real-Time End-to-End Object Detection","date":"2024-05-23","rows_on_this_dataset":6,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/balanced-id-ood-tradeoff-transfer-makes-query","title":"Balanced ID-OOD tradeoff transfer makes query based detectors good few shot learners","date":"2024-05-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/isearle-improving-textual-inversion-for-zero","title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval","date":"2024-05-05","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/3shnet-boosting-image-sentence-retrieval-via","title":"3SHNet: Boosting Image-Sentence Retrieval via Visual Semantic-Spatial Self-Highlighting","date":"2024-04-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":10,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dynamic-self-adaptive-multiscale-distillation","title":"Dynamic Self-adaptive Multiscale Distillation from Pre-trained Multimodal Large Model for Efficient Cross-modal Representation Learning","date":"2024-04-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/yolov5-6d-advancing-6-dof-instrument-pose","title":"YOLOv5-6D: Advancing 6-DoF Instrument Pose Estimation in Variable X-Ray Imaging Geometries","date":"2024-03-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vit-comer-vision-transformer-with","title":"ViT-CoMer: Vision Transformer with Convolutional Multi-scale Feature Interaction for Dense Predictions","date":"2024-03-13","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/weaksam-segment-anything-meets-weakly","title":"WeakSAM: Segment Anything Meets Weakly-supervised Instance-level Recognition","date":"2024-02-22","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/yolov9-learning-what-you-want-to-learn-using","title":"YOLOv9: Learning What You Want to Learn Using Programmable Gradient Information","date":"2024-02-21","rows_on_this_dataset":8,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":32,"samples_ran":21,"samples_unverified":11,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/complete-instances-mining-for-weakly-1","title":"Complete Instances Mining for Weakly Supervised Instance Segmentation","date":"2024-02-12","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":8,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-domain-few-shot-object-detection-via","title":"Cross-Domain Few-Shot Object Detection via Enhanced Open-Set Object Detector","date":"2024-02-05","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":9,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/boldsymbol-m-2-encoder-advancing-bilingual","title":"M2-Encoder: Advancing Bilingual Image-Text Understanding by Large-scale Efficient Pretraining","date":"2024-01-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/umg-clip-a-unified-multi-granularity-vision","title":"UMG-CLIP: A Unified Multi-Granularity Vision Generalist for Open-World Understanding","date":"2024-01-12","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/mst-adaptive-multi-scale-tokens-guided","title":"MST: Adaptive Multi-Scale Tokens Guided Interactive Segmentation","date":"2024-01-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-diffusion-based-image-synthesis-1","title":"Improving Diffusion-Based Image Synthesis with Context Prediction","date":"2024-01-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/harnessing-diffusion-models-for-visual","title":"Harnessing Diffusion Models for Visual Perception with Meta Prompts","date":"2023-12-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internvl-scaling-up-vision-foundation-models","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","date":"2023-12-21","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/general-object-foundation-model-for-images","title":"General Object Foundation Model for Images and Videos at Scale","date":"2023-12-14","rows_on_this_dataset":12,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":8,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/maskconver-revisiting-pure-convolution-model","title":"MaskConver: Revisiting Pure Convolution Model for Panoptic Segmentation","date":"2023-12-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/lyrics-boosting-fine-grained-language-vision","title":"Lyrics: Boosting Fine-grained Language-Vision Alignment and Comprehension via Semantic-aware Visual Objects","date":"2023-12-08","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hulk-a-universal-knowledge-translator-for","title":"Hulk: A Universal Knowledge Translator for Human-Centric Tasks","date":"2023-12-04","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":12,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/transnext-robust-foveal-visual-perception-for","title":"TransNeXt: Robust Foveal Visual Perception for Vision Transformers","date":"2023-11-28","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":5,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/on-the-calibration-of-human-pose-estimation","title":"On the Calibration of Human Pose Estimation","date":"2023-11-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/unireplknet-a-universal-perception-large","title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","date":"2023-11-27","rows_on_this_dataset":6,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/neusyre-neuro-symbolic-visual-understanding","title":"NeuSyRE: Neuro-Symbolic Visual Understanding and Reasoning Framework based on Scene Graph Enrichment","date":"2023-11-05","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/re-scoring-using-image-language-similarity","title":"Re-Scoring Using Image-Language Similarity for Few-Shot Object Detection","date":"2023-11-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/unipose-detecting-any-keypoints","title":"X-Pose: Detecting Any Keypoints","date":"2023-10-12","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":11,"samples_unverified":2,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/seeds-semantic-separable-diffusion","title":"SeeDS: Semantic Separable Diffusion Synthesizer for Zero-shot Food Detection","date":"2023-10-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/kandinsky-an-improved-text-to-image-synthesis","title":"Kandinsky: an Improved Text-to-Image Synthesis with Image Prior and Latent Diffusion","date":"2023-10-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":12,"samples_unverified":6,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/context-i2w-mapping-images-to-context","title":"Context-I2W: Mapping Images to Context-dependent Words for Accurate Zero-Shot Composed Image Retrieval","date":"2023-09-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mocae-mixture-of-calibrated-experts","title":"MoCaE: Mixture of Calibrated Experts Significantly Improves Object Detection","date":"2023-09-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/detect-every-thing-with-few-examples","title":"Detect Everything with Few Examples","date":"2023-09-22","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":15,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/implicit-differentiable-outlier-detection","title":"Implicit Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-09-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dat-spatially-dynamic-vision-transformer-with","title":"DAT++: Spatially Dynamic Vision Transformer with Deformable Attention","date":"2023-09-04","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":2,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gkgnet-group-k-nearest-neighbor-based-graph","title":"GKGNet: Group K-Nearest Neighbor based Graph Convolutional Network for Multi-Label Image Recognition","date":"2023-08-28","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/mask-frozen-detr-high-quality-instance","title":"Mask Frozen-DETR: High Quality Instance Segmentation with One GPU","date":"2023-08-07","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/semantic-aware-dual-contrastive-learning-for","title":"Semantic-Aware Dual Contrastive Learning for Multi-label Image Classification","date":"2023-07-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hierarchical-open-vocabulary-universal-image-1","title":"Hierarchical Open-vocabulary Universal Image Segmentation","date":"2023-07-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rethinking-pose-estimation-in-crowds","title":"Rethinking pose estimation in crowds: overcoming the detection information-bottleneck and ambiguity","date":"2023-06-13","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hiera-a-hierarchical-vision-transformer","title":"Hiera: A Hierarchical Vision Transformer without the Bells-and-Whistles","date":"2023-06-01","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":42,"samples_ran":15,"samples_unverified":27,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/raphael-text-to-image-generation-via-large","title":"RAPHAEL: Text-to-Image Generation via Large Mixture of Diffusion Paths","date":"2023-05-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/one-peace-exploring-one-general","title":"ONE-PEACE: Exploring One General Representation Model Toward Unlimited Modalities","date":"2023-05-18","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":2,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/region-aware-pretraining-for-open-vocabulary","title":"Region-Aware Pretraining for Open-Vocabulary Object Detection with Vision Transformers","date":"2023-05-11","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/understanding-gaussian-attention-bias-of","title":"Understanding Gaussian Attention Bias of Vision Transformers Using Effective Receptive Fields","date":"2023-05-08","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tr0n-translator-networks-for-0-shot-plug-and","title":"TR0N: Translator Networks for 0-Shot Plug-and-Play Conditional Generation","date":"2023-04-26","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":9,"samples_unverified":10,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-strong-and-reproducible-object-detector","title":"A Strong and Reproducible Object Detector with Only Public Datasets","date":"2023-04-25","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/valor-vision-audio-language-omni-perception","title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset","date":"2023-04-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/detrs-beat-yolos-on-real-time-object","title":"DETRs Beat YOLOs on Real-time Object Detection","date":"2023-04-17","rows_on_this_dataset":4,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/yolo-drone-airborne-real-time-detection-of","title":"YOLO-Drone:Airborne real-time detection of dense small objects from high-altitude perspective","date":"2023-04-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/layoutdiffusion-controllable-diffusion-model","title":"LayoutDiffusion: Controllable Diffusion Model for Layout-to-image Generation","date":"2023-03-30","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":19,"samples_ran":5,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/beyond-appearance-a-semantic-controllable","title":"Beyond Appearance: a Semantic Controllable Self-Supervised Learning Framework for Human-Centric Visual Tasks","date":"2023-03-30","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mammut-a-simple-architecture-for-joint","title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","date":"2023-03-29","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zero-shot-composed-image-retrieval-with","title":"Zero-Shot Composed Image Retrieval with Textual Inversion","date":"2023-03-27","rows_on_this_dataset":4,"code_links":2,"syntology":null},{"paper":"/paper/plug-and-play-regulators-for-image-text","title":"Plug-and-Play Regulators for Image-Text Matching","date":"2023-03-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/human-pose-as-compositional-tokens","title":"Human Pose as Compositional Tokens","date":"2023-03-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/biformer-vision-transformer-with-bi-level","title":"BiFormer: Vision Transformer with Bi-Level Routing Attention","date":"2023-03-15","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":6,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-simple-framework-for-open-vocabulary","title":"A Simple Framework for Open-Vocabulary Segmentation and Detection","date":"2023-03-14","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/universal-instance-perception-as-object","title":"Universal Instance Perception as Object Discovery and Retrieval","date":"2023-03-12","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/humanbench-towards-general-human-centric","title":"HumanBench: Towards General Human-centric Perception with Projector Assisted Pretraining","date":"2023-03-10","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":15,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scaling-up-gans-for-text-to-image-synthesis","title":"Scaling up GANs for Text-to-Image Synthesis","date":"2023-03-09","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":8,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/grounding-dino-marrying-dino-with-grounded","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","date":"2023-03-09","rows_on_this_dataset":2,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unihcp-a-unified-model-for-human-centric","title":"UniHCP: A Unified Model for Human-Centric Perceptions","date":"2023-03-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":7,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/causality-compensated-attention-for","title":"Causality Compensated Attention for Contextual Biased Visual Recognition","date":"2023-02-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/on-the-ideal-number-of-groups-for-isometric","title":"On the Ideal Number of Groups for Isometric Gradient Propagation","date":"2023-02-07","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/pic2word-mapping-pictures-to-words-for-zero","title":"Pic2Word: Mapping Pictures to Words for Zero-shot Composed Image Retrieval","date":"2023-02-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/galip-generative-adversarial-clips-for-text","title":"GALIP: Generative Adversarial CLIPs for Text-to-Image Synthesis","date":"2023-01-30","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":7,"samples_unverified":4,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","rows_on_this_dataset":4,"code_links":17,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/simple-diffusion-end-to-end-diffusion-for","title":"Simple diffusion: End-to-end diffusion for high resolution images","date":"2023-01-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/stylegan-t-unlocking-the-power-of-gans-for","title":"StyleGAN-T: Unlocking the Power of GANs for Fast Large-Scale Text-to-Image Synthesis","date":"2023-01-23","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/gligen-open-set-grounded-text-to-image","title":"GLIGEN: Open-Set Grounded Text-to-Image Generation","date":"2023-01-17","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/yolov6-v3-0-a-full-scale-reloading","title":"YOLOv6 v3.0: A Full-Scale Reloading","date":"2023-01-13","rows_on_this_dataset":6,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/toward-building-general-foundation-models-for","title":"Toward Building General Foundation Models for Language, Vision, and Vision-Language Understanding Tasks","date":"2023-01-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/napreg-nouns-as-proxies-regularization-for","title":"NAPReg: Nouns As Proxies Regularization for Semantically Aware Cross-Modal Embeddings","date":"2023-01-07","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/muse-text-to-image-generation-via-masked","title":"Muse: Text-To-Image Generation via Masked Generative Transformers","date":"2023-01-02","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":18,"samples_unverified":3,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/detr-does-not-need-multi-scale-or-locality","title":"DETR Does Not Need Multi-Scale or Locality Design","date":"2023-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/reversible-column-networks","title":"Reversible Column Networks","date":"2022-12-22","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":8,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/position-guided-text-prompt-for-vision","title":"Position-guided Text Prompt for Vision-Language Pre-training","date":"2022-12-19","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rtmdet-an-empirical-study-of-designing-real","title":"RTMDet: An Empirical Study of Designing Real-Time Object Detectors","date":"2022-12-14","rows_on_this_dataset":2,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":3,"samples_unverified":17,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/resolving-semantic-confusions-for-improved-1","title":"Resolving Semantic Confusions for Improved Zero-Shot Detection","date":"2022-12-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/nms-strikes-back","title":"NMS Strikes Back","date":"2022-12-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deepcut-unsupervised-segmentation-using-graph","title":"DeepCut: Unsupervised Segmentation using Graph Neural Networks Clustering","date":"2022-12-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/x-paste-revisit-copy-paste-at-scale-with-clip","title":"X-Paste: Revisiting Scalable Copy-Paste for Instance Segmentation using CLIP and StableDiffusion","date":"2022-12-07","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/diffusioninst-diffusion-model-for-instance","title":"DiffusionInst: Diffusion Model for Instance Segmentation","date":"2022-12-06","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":11,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/box2mask-box-supervised-instance-segmentation","title":"Box2Mask: Box-supervised Instance Segmentation via Level-set Evolution","date":"2022-12-03","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/grit-a-generative-region-to-text-transformer","title":"GRiT: A Generative Region-to-text Transformer for Object Understanding","date":"2022-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/segclip-patch-aggregation-with-learnable","title":"SegCLIP: Patch Aggregation with Learnable Centers for Open-Vocabulary Semantic Segmentation","date":"2022-11-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rethinking-alignment-and-uniformity-in","title":"Rethinking Alignment and Uniformity in Unsupervised Semantic Segmentation","date":"2022-11-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/shifted-diffusion-for-text-to-image","title":"Shifted Diffusion for Text-to-image Generation","date":"2022-11-24","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":7,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/damo-yolo-a-report-on-real-time-object","title":"DAMO-YOLO : A Report on Real-Time Object Detection Design","date":"2022-11-23","rows_on_this_dataset":4,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":1,"samples_unverified":11,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":2,"samples_unverified":4,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/retrieval-augmented-multimodal-language","title":"Retrieval-Augmented Multimodal Language Modeling","date":"2022-11-22","rows_on_this_dataset":13,"code_links":0,"syntology":null},{"paper":"/paper/detrs-with-collaborative-hybrid-assignments","title":"DETRs with Collaborative Hybrid Assignments Training","date":"2022-11-22","rows_on_this_dataset":6,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-all-in-one-pre-training-via","title":"Towards All-in-one Pre-training via Maximizing Multi-modal Mutual Information","date":"2022-11-17","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/eva-exploring-the-limits-of-masked-visual","title":"EVA: Exploring the Limits of Masked Visual Representation Learning at Scale","date":"2022-11-14","rows_on_this_dataset":4,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/oneformer-one-transformer-to-rule-universal","title":"OneFormer: One Transformer to Rule Universal Image Segmentation","date":"2022-11-10","rows_on_this_dataset":6,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internimage-exploring-large-scale-vision","title":"InternImage: Exploring Large-Scale Vision Foundation Models with Deformable Convolutions","date":"2022-11-10","rows_on_this_dataset":11,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/group-detr-v2-strong-object-detector-with-1","title":"Group DETR v2: Strong Object Detector with Encoder-Decoder Pretraining","date":"2022-11-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/efficient-multi-order-gated-aggregation","title":"MogaNet: Multi-order Gated Aggregation Network","date":"2022-11-07","rows_on_this_dataset":10,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":12,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/could-giant-pretrained-image-models-extract","title":"Could Giant Pretrained Image Models Extract Universal Representations?","date":"2022-11-03","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/ediffi-text-to-image-diffusion-models-with-an","title":"eDiff-I: Text-to-Image Diffusion Models with an Ensemble of Expert Denoisers","date":"2022-11-02","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":4,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ernie-vilg-2-0-improving-text-to-image","title":"ERNIE-ViLG 2.0: Improving Text-to-Image Diffusion Model with Knowledge-Enhanced Mixture-of-Denoising-Experts","date":"2022-10-27","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/unsupervised-image-semantic-segmentation","title":"Unsupervised Image Semantic Segmentation through Superpixels and Graph Neural Networks","date":"2022-10-21","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dissecting-deep-metric-learning-losses-for","title":"Dissecting Deep Metric Learning Losses for Image-Text Retrieval","date":"2022-10-21","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/towards-sustainable-self-supervised-learning","title":"Towards Sustainable Self-supervised Learning","date":"2022-10-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":6,"samples_unverified":2,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/swinv2-imagen-hierarchical-vision-transformer","title":"Swinv2-Imagen: Hierarchical Vision Transformer Diffusion Models for Text-to-Image Generation","date":"2022-10-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/perceptual-grouping-in-vision-language-models","title":"Perceptual Grouping in Contrastive Vision-Language Models","date":"2022-10-18","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/a-tri-layer-plugin-to-improve-occluded","title":"A Tri-Layer Plugin to Improve Occluded Detection","date":"2022-10-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/move-unsupervised-movable-object-segmentation","title":"MOVE: Unsupervised Movable Object Segmentation and Detection","date":"2022-10-14","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-discriminative-and-transferable-one","title":"Towards Discriminative and Transferable One-Stage Few-Shot Object Detectors","date":"2022-10-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/boxteacher-exploring-high-quality-pseudo","title":"BoxTeacher: Exploring High-Quality Pseudo Labels for Weakly Supervised Instance Segmentation","date":"2022-10-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/moat-alternating-mobile-convolution-and","title":"MOAT: Alternating Mobile Convolution and Attention Brings Strong Vision Models","date":"2022-10-04","rows_on_this_dataset":18,"code_links":2,"syntology":null},{"paper":"/paper/k-means-for-unsupervised-instance","title":"K-means for unsupervised instance segmentation using a self-supervised transformer","date":"2022-10-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ernie-vil-2-0-multi-view-contrastive-learning","title":"ERNIE-ViL 2.0: Multi-view Contrastive Learning for Image-Text Pre-training","date":"2022-09-30","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/re-imagen-retrieval-augmented-text-to-image","title":"Re-Imagen: Retrieval-Augmented Text-to-Image Generator","date":"2022-09-29","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/dilated-neighborhood-attention-transformer","title":"Dilated Neighborhood Attention Transformer","date":"2022-09-29","rows_on_this_dataset":2,"code_links":7,"syntology":null},{"paper":"/paper/all-are-worth-words-a-vit-backbone-for-score","title":"All are Worth Words: A ViT Backbone for Diffusion Models","date":"2022-09-25","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/omnivl-one-foundation-model-for-image","title":"OmniVL:One Foundation Model for Image-Language and Video-Language Tasks","date":"2022-09-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/combining-metric-learning-and-attention-heads","title":"Combining Metric Learning and Attention Heads For Accurate and Efficient Multilabel Image Classification","date":"2022-09-14","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/exploring-target-representations-for-masked","title":"Exploring Target Representations for Masked Autoencoders","date":"2022-09-08","rows_on_this_dataset":8,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":5,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/yolov6-a-single-stage-object-detection","title":"YOLOv6: A Single-Stage Object Detection Framework for Industrial Applications","date":"2022-09-07","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dpit-dual-pipeline-integrated-transformer-for","title":"DPIT: Dual-Pipeline Integrated Transformer for Human Pose Estimation","date":"2022-09-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/gswin-gated-mlp-vision-model-with","title":"gSwin: Gated MLP Vision Model with Hierarchical Structure of Shifted Window","date":"2022-08-24","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/image-as-a-foreign-language-beit-pretraining","title":"Image as a Foreign Language: BEiT Pretraining for All Vision and Vision-Language Tasks","date":"2022-08-22","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/a-dual-modality-approach-for-zero-shot-multi","title":"Open Vocabulary Multi-Label Classification with Dual-Modal Decoder on Aligned Visual-Textual Features","date":"2022-08-19","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/hierarchical-attention-network-for-few-shot","title":"Hierarchical Attention Network for Few-Shot Object Detection via Meta-Contrastive Learning","date":"2022-08-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/expansionnet-v2-block-static-expansion-in","title":"Exploiting Multiple Sequence Lengths in Fast End to End Training for Image Captioning","date":"2022-08-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/analog-bits-generating-discrete-data-using","title":"Analog Bits: Generating Discrete Data using Diffusion Models with Self-Conditioning","date":"2022-08-08","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":13,"samples_unverified":7,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/aladin-distilling-fine-grained-alignment","title":"ALADIN: Distilling Fine-grained Alignment Scores for Efficient Image-Text Matching and Retrieval","date":"2022-07-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hornet-efficient-high-order-spatial","title":"HorNet: Efficient High-Order Spatial Interactions with Recursive Gated Convolutions","date":"2022-07-28","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/k-means-mask-transformer","title":"kMaX-DeepLab: k-means Mask Transformer","date":"2022-07-08","rows_on_this_dataset":4,"code_links":3,"syntology":null},{"paper":"/paper/yolov7-trainable-bag-of-freebies-sets-new","title":"YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors","date":"2022-07-06","rows_on_this_dataset":10,"code_links":21,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":1,"samples_unverified":10,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-constrained-inference-optimization-on","title":"Self-Constrained Inference Optimization on Structural Groups for Human Pose Estimation","date":"2022-07-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/boosting-r-cnn-reweighting-r-cnn-samples-by","title":"Boosting R-CNN: Reweighting R-CNN Samples by RPN's Error for Underwater Object Detection","date":"2022-06-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/i-2r-net-intra-and-inter-human-relation","title":"I^2R-Net: Intra- and Inter-Human Relation Network for Multi-Person Pose Estimation","date":"2022-06-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/0-1-deep-neural-networks-via-block-coordinate","title":"0/1 Deep Neural Networks via Block Coordinate Descent","date":"2022-06-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cmt-deeplab-clustering-mask-transformers-for-1","title":"CMT-DeepLab: Clustering Mask Transformers for Panoptic Segmentation","date":"2022-06-17","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deep-multi-task-networks-for-occluded","title":"Deep Multi-Task Networks For Occluded Pedestrian Pose Estimation","date":"2022-06-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/glipv2-unifying-localization-and-vision","title":"GLIPv2: Unifying Localization and Vision-Language Understanding","date":"2022-06-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mask-dino-towards-a-unified-transformer-based-1","title":"Mask DINO: Towards A Unified Transformer-based Framework for Object Detection and Segmentation","date":"2022-06-06","rows_on_this_dataset":6,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":11,"samples_unverified":2,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/contrastive-learning-rivals-masked-image","title":"Contrastive Learning Rivals Masked Image Modeling in Fine-tuning via Feature Distillation","date":"2022-05-27","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":2,"samples_unverified":6,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/architecture-agnostic-masked-image-modeling","title":"Architecture-Agnostic Masked Image Modeling -- From ViT back to CNN","date":"2022-05-27","rows_on_this_dataset":4,"code_links":3,"syntology":null},{"paper":"/paper/revealing-the-dark-secrets-of-masked-image","title":"Revealing the Dark Secrets of Masked Image Modeling","date":"2022-05-26","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mixmim-mixed-and-masked-image-modeling-for","title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","date":"2022-05-26","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/photorealistic-text-to-image-diffusion-models","title":"Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding","date":"2022-05-23","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":5,"samples_unverified":6,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/uniform-masking-enabling-mae-pre-training-for","title":"Uniform Masking: Enabling MAE Pre-training for Pyramid-based Vision Transformers with Locality","date":"2022-05-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/integral-migrating-pre-trained-transformer","title":"Integrally Migrating Pre-trained Transformer Encoder-decoders for Visual Object Detection","date":"2022-05-19","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/vision-transformer-adapter-for-dense","title":"Vision Transformer Adapter for Dense Predictions","date":"2022-05-17","rows_on_this_dataset":11,"code_links":2,"syntology":null},{"paper":"/paper/simple-open-vocabulary-object-detection-with","title":"Simple Open-Vocabulary Object Detection with Vision Transformers","date":"2022-05-12","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/aggpose-deep-aggregation-vision-transformer","title":"AggPose: Deep Aggregation Vision Transformer for Infant Pose Estimation","date":"2022-05-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":9,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lite-pose-efficient-architecture-design-for","title":"Lite Pose: Efficient Architecture Design for 2D Human Pose Estimation","date":"2022-05-03","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/flamingo-a-visual-language-model-for-few-shot-1","title":"Flamingo: a Visual Language Model for Few-Shot Learning","date":"2022-04-29","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":18,"samples_unverified":6,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cogview2-faster-and-better-text-to-image","title":"CogView2: Faster and Better Text-to-Image Generation via Hierarchical Transformers","date":"2022-04-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vitpose-simple-vision-transformer-baselines","title":"ViTPose: Simple Vision Transformer Baselines for Human Pose Estimation","date":"2022-04-26","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":31,"samples_ran":18,"samples_unverified":13,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/understanding-the-robustness-in-vision","title":"Understanding The Robustness in Vision Transformers","date":"2022-04-26","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/recurrent-affine-transformation-for-text-to","title":"Recurrent Affine Transformation for Text-to-image Synthesis","date":"2022-04-22","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/dite-hrnet-dynamic-lightweight-high","title":"Dite-HRNet: Dynamic Lightweight High-Resolution Network for Human Pose Estimation","date":"2022-04-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/centernet-for-object-detection","title":"CenterNet++ for Object Detection","date":"2022-04-18","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/yolo-pose-enhancing-yolo-for-multi-person","title":"YOLO-Pose: Enhancing YOLO for Multi Person Pose Estimation Using Object Keypoint Similarity Loss","date":"2022-04-14","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/hierarchical-text-conditional-image","title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","date":"2022-04-13","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":38,"samples_ran":29,"samples_unverified":9,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cfa-constraint-based-finetuning-approach-for","title":"CFA: Constraint-based Finetuning Approach for Generalized Few-Shot Object Detection","date":"2022-04-11","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/davit-dual-attention-vision-transformers","title":"DaViT: Dual Attention Vision Transformers","date":"2022-04-07","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":8,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/knn-diffusion-image-generation-via-large","title":"KNN-Diffusion: Image Generation via Large-Scale Retrieval","date":"2022-04-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/maxvit-multi-axis-vision-transformer","title":"MaxViT: Multi-Axis Vision Transformer","date":"2022-04-04","rows_on_this_dataset":3,"code_links":15,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":53,"samples_ran":33,"samples_unverified":20,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vista-vision-and-scene-text-aggregation-for","title":"ViSTA: Vision and Scene Text Aggregation for Cross-Modal Retrieval","date":"2022-03-31","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/pp-yoloe-an-evolved-version-of-yolo","title":"PP-YOLOE: An evolved version of YOLO","date":"2022-03-30","rows_on_this_dataset":9,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":27,"samples_ran":5,"samples_unverified":22,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploring-plain-vision-transformer-backbones","title":"Exploring Plain Vision Transformer Backbones for Object Detection","date":"2022-03-30","rows_on_this_dataset":4,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/make-a-scene-scene-based-text-to-image","title":"Make-A-Scene: Scene-Based Text-to-Image Generation with Human Priors","date":"2022-03-24","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/focal-modulation-networks","title":"Focal Modulation Networks","date":"2022-03-22","rows_on_this_dataset":5,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/activemlp-an-mlp-like-architecture-with","title":"Active Token Mixer","date":"2022-03-11","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/e2ec-an-end-to-end-contour-based-method-for","title":"E2EC: An End-to-End Contour-based Method for High-Quality High-Speed Instance Segmentation","date":"2022-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":8,"samples_unverified":2,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dino-detr-with-improved-denoising-anchor-1","title":"DINO: DETR with Improved DeNoising Anchor Boxes for End-to-End Object Detection","date":"2022-03-07","rows_on_this_dataset":4,"code_links":16,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":7,"samples_unverified":8,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lile-look-in-depth-before-looking-elsewhere-a","title":"LILE: Look In-Depth before Looking Elsewhere -- A Dual Attention Network using Transformers for Cross-Modal Information Retrieval in Histopathology Archives","date":"2022-03-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dn-detr-accelerate-detr-training-by","title":"DN-DETR: Accelerate DETR Training by Introducing Query DeNoising","date":"2022-03-02","rows_on_this_dataset":1,"code_links":17,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":22,"samples_ran":15,"samples_unverified":7,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/self-supervised-transformers-for-unsupervised","title":"Self-Supervised Transformers for Unsupervised Object Discovery using Normalized Cut","date":"2022-02-23","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/isda-position-aware-instance-segmentation","title":"ISDA: Position-Aware Instance Segmentation with Deformable Attention","date":"2022-02-23","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/vision-language-pre-training-with-triple","title":"Vision-Language Pre-Training with Triple Contrastive Learning","date":"2022-02-21","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/visual-attention-network","title":"Visual Attention Network","date":"2022-02-20","rows_on_this_dataset":2,"code_links":21,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/truncated-diffusion-probabilistic-models","title":"Truncated Diffusion Probabilistic Models and Diffusion-based Adversarial Auto-Encoders","date":"2022-02-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/from-node-to-graph-joint-reasoning-on-visual","title":"From Node to Graph: Joint Reasoning on Visual-Semantic Relational Graph for Zero-Shot Detection","date":"2022-02-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/context-autoencoder-for-self-supervised","title":"Context Autoencoder for Self-Supervised Representation Learning","date":"2022-02-07","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/the-devil-is-in-the-labels-semantic","title":"Scaling up Multi-domain Semantic Segmentation with Sentence Embeddings","date":"2022-02-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dab-detr-dynamic-anchor-boxes-are-better-1","title":"DAB-DETR: Dynamic Anchor Boxes are Better Queries for DETR","date":"2022-01-28","rows_on_this_dataset":2,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":6,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/when-shift-operation-meets-vision-transformer","title":"When Shift Operation Meets Vision Transformer: An Extremely Simple Alternative to Attention Mechanism","date":"2022-01-26","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/poseur-direct-human-pose-regression-with","title":"Poseur: Direct Human Pose Regression with Transformers","date":"2022-01-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vision-transformer-with-deformable-attention","title":"Vision Transformer with Deformable Attention","date":"2022-01-03","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":8,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/robust-region-feature-synthesizer-for-zero","title":"Robust Region Feature Synthesizer for Zero-Shot Object Detection","date":"2022-01-01","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/contextual-debiasing-for-visual-recognition","title":"Contextual Debiasing for Visual Recognition With Causal Mechanisms","date":"2022-01-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/ernie-vilg-unified-generative-pre-training","title":"ERNIE-ViLG: Unified Generative Pre-training for Bidirectional Vision-Language Generation","date":"2021-12-31","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/augmenting-convolutional-networks-with","title":"Augmenting Convolutional networks with attention-based aggregation","date":"2021-12-27","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/elsa-enhanced-local-self-attention-for-vision","title":"ELSA: Enhanced Local Self-Attention for Vision Transformer","date":"2021-12-23","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mpvit-multi-path-vision-transformer-for-dense","title":"MPViT: Multi-Path Vision Transformer for Dense Prediction","date":"2021-12-21","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/high-resolution-image-synthesis-with-latent","title":"High-Resolution Image Synthesis with Latent Diffusion Models","date":"2021-12-20","rows_on_this_dataset":3,"code_links":41,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":28,"samples_ran":19,"samples_unverified":9,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/glide-towards-photorealistic-image-generation","title":"GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models","date":"2021-12-20","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":9,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bapose-bottom-up-pose-estimation-with","title":"BAPose: Bottom-Up Pose Estimation with Disentangled Waterfall Representations","date":"2021-12-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/recurrent-glimpse-based-decoder-for-detection","title":"Recurrent Glimpse-based Decoder for Detection with Transformer","date":"2021-12-09","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/pe-former-pose-estimation-transformer","title":"PE-former: Pose Estimation Transformer","date":"2021-12-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/flava-a-foundational-language-and-vision","title":"FLAVA: A Foundational Language And Vision Alignment Model","date":"2021-12-08","rows_on_this_dataset":3,"code_links":4,"syntology":null},{"paper":"/paper/grounded-language-image-pre-training","title":"Grounded Language-Image Pre-training","date":"2021-12-07","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/masked-attention-mask-transformer-for","title":"Masked-attention Mask Transformer for Universal Image Segmentation","date":"2021-12-02","rows_on_this_dataset":6,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":2,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/improved-multiscale-vision-transformers-for","title":"MViTv2: Improved Multiscale Vision Transformers for Classification and Detection","date":"2021-12-02","rows_on_this_dataset":8,"code_links":9,"syntology":null},{"paper":"/paper/fusedream-training-free-text-to-image","title":"FuseDream: Training-Free Text-to-Image Generation with Improved CLIP+GAN Space Optimization","date":"2021-12-02","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/vector-quantized-diffusion-model-for-text-to","title":"Vector Quantized Diffusion Model for Text-to-Image Synthesis","date":"2021-11-29","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/lafite-towards-language-free-training-for","title":"LAFITE: Towards Language-Free Training for Text-to-Image Generation","date":"2021-11-27","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":4,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/revisiting-efficient-object-detection","title":"MAE-DET: Revisiting Maximum Entropy Principle in Zero-Shot NAS for Efficient Object Detection","date":"2021-11-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mask-transfiner-for-high-quality-instance","title":"Mask Transfiner for High-Quality Instance Segmentation","date":"2021-11-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ml-decoder-scalable-and-versatile","title":"ML-Decoder: Scalable and Versatile Classification Head","date":"2021-11-25","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attend-to-who-you-are-supervising-self-1","title":"Attend to Who You Are: Supervising Self-Attention for Keypoint Detection and Instance-Aware Association","date":"2021-11-25","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/nuwa-visual-synthesis-pre-training-for-neural","title":"NÜWA: Visual Synthesis Pre-training for Neural visUal World creAtion","date":"2021-11-24","rows_on_this_dataset":7,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/focal-and-global-knowledge-distillation-for","title":"Focal and Global Knowledge Distillation for Detectors","date":"2021-11-23","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/multi-modal-transformers-excel-at-class","title":"Class-agnostic Object Detection with Multi-modal Transformer","date":"2021-11-22","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/metaformer-is-actually-what-you-need-for","title":"MetaFormer Is Actually What You Need for Vision","date":"2021-11-22","rows_on_this_dataset":1,"code_links":18,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/l-verse-bidirectional-generation-between","title":"L-Verse: Bidirectional Generation Between Image and Text","date":"2021-11-22","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/florence-a-new-foundation-model-for-computer","title":"Florence: A New Foundation Model for Computer Vision","date":"2021-11-22","rows_on_this_dataset":4,"code_links":2,"syntology":null},{"paper":"/paper/swin-transformer-v2-scaling-up-capacity-and","title":"Swin Transformer V2: Scaling Up Capacity and Resolution","date":"2021-11-18","rows_on_this_dataset":4,"code_links":23,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":30,"samples_ran":3,"samples_unverified":27,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rethinking-keypoint-representations-modeling","title":"Rethinking Keypoint Representations: Modeling Keypoints and Poses as Objects for Multi-Person Human Pose Estimation","date":"2021-11-16","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ibot-image-bert-pre-training-with-online","title":"iBOT: Image BERT Pre-Training with Online Tokenizer","date":"2021-11-15","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/masked-autoencoders-are-scalable-vision","title":"Masked Autoencoders Are Scalable Vision Learners","date":"2021-11-11","rows_on_this_dataset":2,"code_links":58,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":137,"samples_ran":71,"samples_unverified":66,"pointer_only_for_licence":73,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/an-empirical-study-of-training-end-to-end","title":"An Empirical Study of Training End-to-End Vision-and-Language Transformers","date":"2021-11-03","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/plug-and-play-few-shot-object-detection-with","title":"Instant Response Few-shot Object Detection with Meta Strategy and Explicit Localization Inference","date":"2021-10-26","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hrformer-high-resolution-transformer-for","title":"HRFormer: High-Resolution Transformer for Dense Prediction","date":"2021-10-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":7,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-center-of-attention-center-keypoint-1","title":"The Center of Attention: Center-Keypoint Grouping via Attention for Multi-Person Pose Estimation","date":"2021-10-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/transformer-based-dual-relation-graph-for-1","title":"Transformer-based Dual Relation Graph for Multi-label Image Recognition","date":"2021-10-10","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":9,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/infoseg-unsupervised-semantic-image","title":"InfoSeg: Unsupervised Semantic Image Segmentation with Mutual Information Maximization","date":"2021-10-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/m3tr-multi-modal-multi-label-recognition-with","title":"M3TR: Multi-modal Multi-label Recognition with Transformer","date":"2021-10-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/localizing-objects-with-self-supervised","title":"Localizing Objects with Self-Supervised Transformers and no Labels","date":"2021-09-29","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/pix2seq-a-language-modeling-framework-for","title":"Pix2seq: A Language Modeling Framework for Object Detection","date":"2021-09-22","rows_on_this_dataset":6,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":10,"samples_unverified":2,"pointer_only_for_licence":10,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/beyond-semantic-to-instance-segmentation","title":"Beyond Semantic to Instance Segmentation: Weakly-Supervised Instance Segmentation via Semantic Knowledge Transfer and Self-Refinement","date":"2021-09-20","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hptq-hardware-friendly-post-training","title":"HPTQ: Hardware-Friendly Post Training Quantization","date":"2021-09-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/few-shot-object-detection-by-attending-to-per","title":"Few-Shot Object Detection by Attending to Per-Sample-Prototype","date":"2021-09-16","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/anchor-detr-query-design-for-transformer","title":"Anchor DETR: Query Design for Transformer-Based Object Detection","date":"2021-09-15","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/panoptic-segformer","title":"Panoptic SegFormer: Delving Deeper into Panoptic Segmentation with Transformers","date":"2021-09-08","rows_on_this_dataset":6,"code_links":3,"syntology":null},{"paper":"/paper/semantics-guided-contrastive-network-for-zero","title":"Semantics-Guided Contrastive Network for Zero-Shot Object detection","date":"2021-09-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/defrcn-decoupled-faster-r-cnn-for-few-shot","title":"DeFRCN: Decoupled Faster R-CNN for Few-Shot Object Detection","date":"2021-08-20","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/perturb-predict-paraphrase-semi-supervised","title":"Perturb, Predict & Paraphrase: Semi-Supervised Learning using Noisy Student for Image Captioning","date":"2021-08-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tood-task-aligned-one-stage-object-detection","title":"TOOD: Task-aligned One-stage Object Detection","date":"2021-08-17","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":2,"samples_unverified":2,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/conditional-detr-for-fast-training","title":"Conditional DETR for Fast Training Convergence","date":"2021-08-13","rows_on_this_dataset":4,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":5,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/paint-transformer-feed-forward-neural","title":"Paint Transformer: Feed Forward Neural Painting with Stroke Prediction","date":"2021-08-09","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/rethinking-and-improving-relative-position","title":"Rethinking and Improving Relative Position Encoding for Vision Transformer","date":"2021-07-29","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":7,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/query2label-a-simple-transformer-way-to-multi","title":"Query2Label: A Simple Transformer Way to Multi-Label Classification","date":"2021-07-22","rows_on_this_dataset":4,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/inspose-instance-aware-networks-for-single","title":"InsPose: Instance-Aware Networks for Single-Stage Multi-Person Pose Estimation","date":"2021-07-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/yolox-exceeding-yolo-series-in-2021","title":"YOLOX: Exceeding YOLO Series in 2021","date":"2021-07-18","rows_on_this_dataset":4,"code_links":42,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":23,"samples_ran":1,"samples_unverified":22,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/align-before-fuse-vision-and-language","title":"Align before Fuse: Vision and Language Representation Learning with Momentum Distillation","date":"2021-07-16","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/per-pixel-classification-is-not-all-you-need","title":"Per-Pixel Classification is Not All You Need for Semantic Segmentation","date":"2021-07-13","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/greedy-offset-guided-keypoint-grouping-for","title":"Greedy Offset-Guided Keypoint Grouping for Human Pose Estimation","date":"2021-07-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-text-to-image-synthesis-using","title":"Improving Text-to-Image Synthesis Using Contrastive Learning","date":"2021-07-06","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/polarized-self-attention-towards-high-quality-1","title":"Polarized Self-Attention: Towards High-quality Pixel-wise Regression","date":"2021-07-02","rows_on_this_dataset":3,"code_links":6,"syntology":null},{"paper":"/paper/unsupervised-image-segmentation-by-mutual","title":"Unsupervised Image Segmentation by Mutual Information Maximization and Adversarial Regularization","date":"2021-07-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/focal-self-attention-for-local-global","title":"Focal Self-attention for Local-Global Interactions in Vision Transformers","date":"2021-07-01","rows_on_this_dataset":4,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cbnetv2-a-composite-backbone-network","title":"CBNet: A Composite Backbone Network Architecture for Object Detection","date":"2021-07-01","rows_on_this_dataset":9,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/simple-training-strategies-and-model-scaling","title":"Simple Training Strategies and Model Scaling for Object Detection","date":"2021-06-30","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/k-net-towards-unified-image-segmentation","title":"K-Net: Towards Unified Image Segmentation","date":"2021-06-28","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pvtv2-improved-baselines-with-pyramid-vision","title":"PVT v2: Improved Baselines with Pyramid Vision Transformer","date":"2021-06-25","rows_on_this_dataset":1,"code_links":18,"syntology":null},{"paper":"/paper/multi-layered-semantic-representation-network","title":"Multi-layered Semantic Representation Network for Multi-label Image Classification","date":"2021-06-22","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/xcit-cross-covariance-image-transformers","title":"XCiT: Cross-Covariance Image Transformers","date":"2021-06-17","rows_on_this_dataset":4,"code_links":12,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":3,"samples_unverified":11,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-semi-supervised-object-detection","title":"End-to-End Semi-Supervised Object Detection with Soft Teacher","date":"2021-06-16","rows_on_this_dataset":6,"code_links":8,"syntology":null},{"paper":"/paper/dynamic-head-unifying-object-detection-heads","title":"Dynamic Head: Unifying Object Detection Heads with Attentions","date":"2021-06-15","rows_on_this_dataset":9,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":7,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mltr-multi-label-classification-with","title":"MlTr: Multi-label Classification with Transformer","date":"2021-06-11","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/detreg-unsupervised-pretraining-with-region","title":"DETReg: Unsupervised Pretraining with Region Priors for Object Detection","date":"2021-06-08","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/combinatorial-optimization-for-panoptic","title":"Combinatorial Optimization for Panoptic Segmentation: A Fully Differentiable Approach","date":"2021-06-06","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/x-volution-on-the-unification-of-convolution","title":"X-volution: On the unification of convolution and self-attention","date":"2021-06-04","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/solq-segmenting-objects-by-learning-queries","title":"SOLQ: Segmenting Objects by Learning Queries","date":"2021-06-04","rows_on_this_dataset":7,"code_links":1,"syntology":null},{"paper":"/paper/learning-relation-alignment-for-calibrated","title":"Learning Relation Alignment for Calibrated Cross-modal Retrieval","date":"2021-05-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cogview-mastering-text-to-image-generation","title":"CogView: Mastering Text-to-Image Generation via Transformers","date":"2021-05-26","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vipnas-efficient-video-pose-estimation-via","title":"ViPNAS: Efficient Video Pose Estimation via Neural Architecture Search","date":"2021-05-21","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/discobox-weakly-supervised-instance","title":"DiscoBox: Weakly Supervised Instance Segmentation and Semantic Correspondence from Box Supervision","date":"2021-05-13","rows_on_this_dataset":4,"code_links":3,"syntology":null},{"paper":"/paper/you-only-learn-one-representation-unified","title":"You Only Learn One Representation: Unified Network for Multiple Tasks","date":"2021-05-10","rows_on_this_dataset":8,"code_links":9,"syntology":null},{"paper":"/paper/queryinst-parallelly-supervised-mask-query","title":"Instances as Queries","date":"2021-05-05","rows_on_this_dataset":4,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/polarmask-enhanced-polar-representation-for","title":"PolarMask++: Enhanced Polar Representation for Single-Shot Instance Segmentation and Beyond","date":"2021-05-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/istr-end-to-end-instance-segmentation-with","title":"ISTR: End-to-End Instance Segmentation with Transformers","date":"2021-05-03","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/imagenet-21k-pretraining-for-the-masses","title":"ImageNet-21K Pretraining for the Masses","date":"2021-04-22","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lite-hrnet-a-lightweight-high-resolution","title":"Lite-HRNet: A Lightweight High-Resolution Network","date":"2021-04-13","rows_on_this_dataset":2,"code_links":17,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":29,"samples_ran":12,"samples_unverified":17,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/location-sensitive-visual-recognition-with","title":"Location-Sensitive Visual Recognition with Cross-IOU Loss","date":"2021-04-11","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multiple-instance-active-learning-for-object","title":"Multiple instance active learning for object detection","date":"2021-04-06","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/weakly-supervised-instance-segmentation-via","title":"Weakly-supervised Instance Segmentation via Class-agnostic Learning with Salient Images","date":"2021-04-04","rows_on_this_dataset":1,"code_links":0,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/recursively-refined-r-cnn-instance","title":"Recursively Refined R-CNN: Instance Segmentation with Self-RoI Rebalancing","date":"2021-04-03","rows_on_this_dataset":9,"code_links":1,"syntology":null},{"paper":"/paper/2103-15358","title":"Multi-Scale Vision Longformer: A New Vision Transformer for High-Resolution Image Encoding","date":"2021-03-29","rows_on_this_dataset":4,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":6,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/2103-15320","title":"TFPose: Direct Human Pose Estimation with Transformers","date":"2021-03-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/ota-optimal-transport-assignment-for-object","title":"OTA: Optimal Transport Assignment for Object Detection","date":"2021-03-26","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/usb-universal-scale-object-detection","title":"USB: Universal-Scale Object Detection Benchmark","date":"2021-03-25","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/swin-transformer-hierarchical-vision","title":"Swin Transformer: Hierarchical Vision Transformer using Shifted Windows","date":"2021-03-25","rows_on_this_dataset":8,"code_links":80,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":207,"samples_ran":108,"samples_unverified":99,"pointer_only_for_licence":43,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deep-occlusion-aware-instance-segmentation","title":"Deep Occlusion-Aware Instance Segmentation with Overlapping BiLayers","date":"2021-03-23","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":7,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/meta-detr-few-shot-object-detection-via","title":"Meta-DETR: Image-Level Few-Shot Object Detection with Inter-Class Correlation Exploitation","date":"2021-03-22","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/omnipose-a-multi-scale-framework-for-multi","title":"OmniPose: A Multi-Scale Framework for Multi-Person Pose Estimation","date":"2021-03-18","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/you-only-look-one-level-feature","title":"You Only Look One-level Feature","date":"2021-03-17","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/bbam-bounding-box-attribution-map-for-weakly","title":"BBAM: Bounding Box Attribution Map for Weakly Supervised Semantic and Instance Segmentation","date":"2021-03-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/probabilistic-two-stage-detection","title":"Probabilistic two-stage detection","date":"2021-03-12","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/fsce-few-shot-object-detection-via","title":"FSCE: Few-Shot Object Detection via Contrastive Proposal Encoding","date":"2021-03-10","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/beyond-max-margin-class-margin-equilibrium","title":"Beyond Max-Margin: Class Margin Equilibrium for Few-shot Object Detection","date":"2021-03-08","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-open-world-object-detection","title":"Towards Open World Object Detection","date":"2021-03-03","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/openpifpaf-composite-fields-for-semantic","title":"OpenPifPaf: Composite Fields for Semantic Keypoint Detection and Spatio-Temporal Association","date":"2021-03-03","rows_on_this_dataset":2,"code_links":6,"syntology":null},{"paper":"/paper/semantic-relation-reasoning-for-shot-stable","title":"Semantic Relation Reasoning for Shot-Stable Few-Shot Object Detection","date":"2021-03-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/universal-prototype-augmentation-for-few-shot","title":"Universal-Prototype Enhancing for Few-Shot Object Detection","date":"2021-03-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-transferable-visual-models-from","title":"Learning Transferable Visual Models From Natural Language Supervision","date":"2021-02-26","rows_on_this_dataset":2,"code_links":82,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":16,"samples_unverified":4,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/should-i-look-at-the-head-or-the-tail-dual","title":"Dual-Awareness Attention for Few-Shot Object Detection","date":"2021-02-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pyramid-vision-transformer-a-versatile","title":"Pyramid Vision Transformer: A Versatile Backbone for Dense Prediction without Convolutions","date":"2021-02-24","rows_on_this_dataset":2,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":30,"samples_ran":22,"samples_unverified":8,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scaling-up-visual-and-vision-language","title":"Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision","date":"2021-02-11","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":8,"samples_unverified":2,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vilt-vision-and-language-transformer-without","title":"ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision","date":"2021-02-05","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-hypothesis-pose-networks-rethinking-top","title":"Multi-Instance Pose Networks: Rethinking Top-Down Pose Estimation","date":"2021-01-27","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bottleneck-transformers-for-visual","title":"Bottleneck Transformers for Visual Recognition","date":"2021-01-27","rows_on_this_dataset":6,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":49,"samples_ran":26,"samples_unverified":23,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-modal-contrastive-learning-for-text-to","title":"Cross-Modal Contrastive Learning for Text-to-Image Generation","date":"2021-01-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/similarity-reasoning-and-filtration-for-image","title":"Similarity Reasoning and Filtration for Image-Text Matching","date":"2021-01-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":10,"samples_unverified":2,"pointer_only_for_licence":12,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/visualsparta-sparse-transformer-fragment","title":"VisualSparta: An Embarrassingly Simple Approach to Large-scale Text-to-Image Search with Weighted Bag-of-words","date":"2021-01-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/unimo-towards-unified-modal-understanding-and","title":"UNIMO: Towards Unified-Modal Understanding and Generation via Cross-Modal Contrastive Learning","date":"2020-12-31","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/transpose-towards-explainable-human-pose","title":"TransPose: Keypoint Localization via Transformer","date":"2020-12-28","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/global-context-networks","title":"Global Context Networks","date":"2020-12-24","rows_on_this_dataset":4,"code_links":3,"syntology":null},{"paper":"/paper/refine-prediction-fusion-network-for-panoptic","title":"REFINE: Prediction Fusion Network for Panoptic Segmentation","date":"2020-12-15","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/simple-copy-paste-is-a-strong-data","title":"Simple Copy-Paste is a Strong Data Augmentation Method for Instance Segmentation","date":"2020-12-13","rows_on_this_dataset":8,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ada-segment-automated-multi-loss-adaptation","title":"Ada-Segment: Automated Multi-loss Adaptation for Panoptic Segmentation","date":"2020-12-07","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/attention-driven-dynamic-graph-convolutional-1","title":"Attention-Driven Dynamic Graph Convolutional Network for Multi-Label Image Recognition","date":"2020-12-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/parallel-residual-bi-fusion-feature-pyramid","title":"Parallel Residual Bi-Fusion Feature Pyramid Network for Accurate Single-Shot Object Detection","date":"2020-12-03","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/boxinst-high-performance-instance","title":"BoxInst: High-Performance Instance Segmentation with Box Annotations","date":"2020-12-03","rows_on_this_dataset":5,"code_links":2,"syntology":null},{"paper":"/paper/max-deeplab-end-to-end-panoptic-segmentation","title":"MaX-DeepLab: End-to-End Panoptic Segmentation with Mask Transformers","date":"2020-12-01","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/fully-convolutional-networks-for-panoptic","title":"Fully Convolutional Networks for Panoptic Segmentation","date":"2020-12-01","rows_on_this_dataset":4,"code_links":6,"syntology":null},{"paper":"/paper/scalenas-one-shot-learning-of-scale-aware","title":"ScaleNAS: One-Shot Learning of Scale-Aware Representations for Visual Recognition","date":"2020-11-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/torchdistill-a-modular-configuration-driven","title":"torchdistill: A Modular, Configuration-Driven Framework for Knowledge Distillation","date":"2020-11-25","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":26,"samples_ran":10,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/sparse-r-cnn-end-to-end-object-detection-with","title":"Sparse R-CNN: End-to-End Object Detection with Learnable Proposals","date":"2020-11-25","rows_on_this_dataset":4,"code_links":6,"syntology":null},{"paper":"/paper/generalized-focal-loss-v2-learning-reliable","title":"Generalized Focal Loss V2: Learning Reliable Localization Quality Estimation for Dense Object Detection","date":"2020-11-25","rows_on_this_dataset":6,"code_links":5,"syntology":null},{"paper":"/paper/scaling-wide-residual-networks-for-panoptic","title":"Scaling Wide Residual Networks for Panoptic Segmentation","date":"2020-11-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/slender-object-detection-diagnoses-and","title":"Slender Object Detection: Diagnoses and Improvements","date":"2020-11-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/evopose2d-pushing-the-boundaries-of-2d-human","title":"EvoPose2D: Pushing the Boundaries of 2D Human Pose Estimation using Accelerated Neuroevolution with Weight Transfer","date":"2020-11-17","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scaled-yolov4-scaling-cross-stage-partial","title":"Scaled-YOLOv4: Scaling Cross Stage Partial Network","date":"2020-11-16","rows_on_this_dataset":6,"code_links":41,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/in-defense-of-feature-mimicking-for-knowledge","title":"Distilling Knowledge by Mimicking Features","date":"2020-11-03","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":5,"samples_unverified":1,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/relationnet-bridging-visual-representations","title":"RelationNet++: Bridging Visual Representations for Object Detection via Transformer Decoder","date":"2020-10-29","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/activenet-a-computer-vision-based-approach-to","title":"ActiveNet: A computer-vision based approach to determine lethargy","date":"2020-10-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/synthesizing-the-unseen-for-zero-shot-object","title":"Synthesizing the Unseen for Zero-shot Object Detection","date":"2020-10-19","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/background-learnable-cascade-for-zero-shot","title":"Background Learnable Cascade for Zero-Shot Object Detection","date":"2020-10-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deformable-detr-deformable-transformers-for-1","title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","date":"2020-10-08","rows_on_this_dataset":1,"code_links":20,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":55,"samples_ran":29,"samples_unverified":26,"pointer_only_for_licence":21,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/victr-visual-information-captured-text","title":"VICTR: Visual Information Captured Text Representation for Text-to-Image Multimodal Tasks","date":"2020-10-07","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/asymmetric-loss-for-multi-label","title":"Asymmetric Loss For Multi-Label Classification","date":"2020-09-29","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":4,"samples_unverified":8,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-ranking-based-balanced-loss-function","title":"A Ranking-based, Balanced Loss Function Unifying Classification and Localisation in Object Detection","date":"2020-09-28","rows_on_this_dataset":7,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/leveraging-instance-image-and-dataset-level","title":"Leveraging Instance-, Image- and Dataset-Level Information for Weakly Supervised Instance Segmentation","date":"2020-09-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/how-to-train-your-robust-human-pose-estimator","title":"AID: Pushing the Performance Boundary of Human Pose Estimation with Information Dropping Augmentation","date":"2020-08-17","rows_on_this_dataset":6,"code_links":2,"syntology":null},{"paper":"/paper/reducing-label-noise-in-anchor-free-object","title":"Reducing Label Noise in Anchor-Free Object Detection","date":"2020-08-03","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/sipmask-spatial-information-preservation-for","title":"SipMask: Spatial Information Preservation for Fast Image and Video Instance Segmentation","date":"2020-07-29","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/corner-proposal-network-for-anchor-free-two","title":"Corner Proposal Network for Anchor-free, Two-stage Object Detection","date":"2020-07-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/commonality-parsing-network-across-shape-and","title":"Commonality-Parsing Network across Shape and Appearance for Partially Supervised Instance Segmentation","date":"2020-07-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/few-shot-object-detection-and-viewpoint","title":"Few-Shot Object Detection and Viewpoint Estimation for Objects in the Wild","date":"2020-07-23","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":1,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-scale-positive-sample-refinement-for","title":"Multi-Scale Positive Sample Refinement for Few-Shot Object Detection","date":"2020-07-18","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":5,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/reppoints-v2-verification-meets-regression","title":"RepPoints V2: Verification Meets Regression for Object Detection","date":"2020-07-16","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/probabilistic-anchor-assignment-with-iou","title":"Probabilistic Anchor Assignment with IoU Prediction for Object Detection","date":"2020-07-16","rows_on_this_dataset":1,"code_links":5,"syntology":null},{"paper":"/paper/autoregressive-unsupervised-image","title":"Autoregressive Unsupervised Image Segmentation","date":"2020-07-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/toward-unsupervised-multi-object-discovery-in","title":"Toward unsupervised, multi-object discovery in large-scale image collections","date":"2020-07-06","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/houghnet-integrating-near-and-long-range","title":"HoughNet: Integrating near and long-range evidence for bottom-up object detection","date":"2020-07-05","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/multi-label-image-recognition-with-multi","title":"Learning to Discover Multi-Class Attentional Regions for Multi-Label Image Recognition","date":"2020-07-03","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/multi-person-pose-regression-via-pose","title":"SMPR: Single-Stage Multi-Person Pose Regression","date":"2020-06-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/localization-uncertainty-estimation-for","title":"Localization Uncertainty Estimation for Anchor-Free Object Detection","date":"2020-06-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/virtex-learning-visual-representations-from","title":"VirTex: Learning Visual Representations from Textual Annotations","date":"2020-06-11","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rethinking-pre-training-and-self-training","title":"Rethinking Pre-training and Self-training","date":"2020-06-11","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/generalized-focal-loss-learning-qualified-and","title":"Generalized Focal Loss: Learning Qualified and Distributed Bounding Boxes for Dense Object Detection","date":"2020-06-08","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":1,"samples_unverified":23,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/detectors-detecting-objects-with-recursive-1","title":"DetectoRS: Detecting Objects with Recursive Feature Pyramid and Switchable Atrous Convolution","date":"2020-06-03","rows_on_this_dataset":6,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/interactive-object-segmentation-with-inside","title":"Interactive Object Segmentation With Inside-Outside Guidance","date":"2020-06-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/d2det-towards-high-quality-object-detection","title":"D2Det: Towards High Quality Object Detection and Instance Segmentation","date":"2020-06-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/end-to-end-object-detection-with-transformers","title":"End-to-End Object Detection with Transformers","date":"2020-05-26","rows_on_this_dataset":5,"code_links":37,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":92,"samples_ran":59,"samples_unverified":33,"pointer_only_for_licence":19,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attention-guided-context-feature-pyramid","title":"Attention-guided Context Feature Pyramid Network for Object Detection","date":"2020-05-23","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/scale-equalizing-pyramid-convolution-for","title":"Scale-Equalizing Pyramid Convolution for Object Detection","date":"2020-05-06","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/a-novel-region-of-interest-extraction-layer","title":"A novel Region of Interest Extraction Layer for Instance Segmentation","date":"2020-04-28","rows_on_this_dataset":4,"code_links":5,"syntology":null},{"paper":"/paper/yolov4-optimal-speed-and-accuracy-of-object","title":"YOLOv4: Optimal Speed and Accuracy of Object Detection","date":"2020-04-23","rows_on_this_dataset":4,"code_links":223,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":184,"samples_ran":24,"samples_unverified":160,"pointer_only_for_licence":8,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/resnest-split-attention-networks","title":"ResNeSt: Split-Attention Networks","date":"2020-04-19","rows_on_this_dataset":11,"code_links":36,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":48,"samples_ran":8,"samples_unverified":40,"pointer_only_for_licence":23,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/oscar-object-semantics-aligned-pre-training","title":"Oscar: Object-Semantics Aligned Pre-training for Vision-Language Tasks","date":"2020-04-13","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":23,"samples_ran":11,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dynamic-r-cnn-towards-high-quality-object","title":"Dynamic R-CNN: Towards High Quality Object Detection via Dynamic Training","date":"2020-04-13","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":2,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/instance-aware-context-focused-and-memory","title":"Instance-aware, Context-focused, and Memory-efficient Weakly Supervised Object Detection","date":"2020-04-09","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pixel-consensus-voting-for-panoptic","title":"Pixel Consensus Voting for Panoptic Segmentation","date":"2020-04-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/saccadenet-a-fast-and-accurate-object","title":"SaccadeNet: A Fast and Accurate Object Detector","date":"2020-03-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/solov2-dynamic-faster-and-stronger","title":"SOLOv2: Dynamic and Fast Instance Segmentation","date":"2020-03-23","rows_on_this_dataset":1,"code_links":18,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":38,"samples_ran":15,"samples_unverified":23,"pointer_only_for_licence":24,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/epsnet-efficient-panoptic-segmentation","title":"EPSNet: Efficient Panoptic Segmentation Network with Cross-layer Attention Fusion","date":"2020-03-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/revisiting-the-sibling-head-in-object","title":"Revisiting the Sibling Head in Object Detector","date":"2020-03-17","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/axial-deeplab-stand-alone-axial-attention-for","title":"Axial-DeepLab: Stand-Alone Axial-Attention for Panoptic Segmentation","date":"2020-03-17","rows_on_this_dataset":5,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/object-centric-image-generation-from-layouts","title":"Object-Centric Image Generation from Layouts","date":"2020-03-16","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/frustratingly-simple-few-shot-object","title":"Frustratingly Simple Few-Shot Object Detection","date":"2020-03-16","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-delicate-local-representations-for","title":"Learning Delicate Local Representations for Multi-Person Pose Estimation","date":"2020-03-09","rows_on_this_dataset":5,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/imram-iterative-matching-with-recurrent","title":"IMRAM: Iterative Matching with Recurrent Attention Memory for Cross-Modal Image-Text Retrieval","date":"2020-03-08","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-u-net-based-discriminator-for-generative","title":"A U-Net Based Discriminator for Generative Adversarial Networks","date":"2020-02-28","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/cross-iteration-batch-normalization","title":"Cross-Iteration Batch Normalization","date":"2020-02-13","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-high-performance-human-keypoint","title":"Towards High Performance Human Keypoint Detection","date":"2020-02-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/imagebert-cross-modal-pre-training-with-large","title":"ImageBERT: Cross-modal Pre-training with Large-scale Weak-supervised Image-Text Data","date":"2020-01-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/blendmask-top-down-meets-bottom-up-for","title":"BlendMask: Top-Down Meets Bottom-Up for Instance Segmentation","date":"2020-01-02","rows_on_this_dataset":1,"code_links":9,"syntology":null},{"paper":"/paper/multi-label-graph-convolutional-network","title":"Multi-Label Graph Convolutional Network Representation Learning","date":"2019-12-26","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/benchmark-for-generic-product-detection-a","title":"Benchmark for Generic Product Detection: A Low Data Baseline for Dense Object Detection","date":"2019-12-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/m2-meshed-memory-transformer-for-image","title":"Meshed-Memory Transformer for Image Captioning","date":"2019-12-17","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":4,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-modality-attention-with-semantic-graph","title":"Cross-Modality Attention with Semantic Graph Embedding for Multi-Label Classification","date":"2019-12-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-canonical-representations-for-scene","title":"Learning Canonical Representations for Scene Graph to Image Generation","date":"2019-12-16","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":2,"samples_unverified":15,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rdsnet-a-new-deep-architecture-for-reciprocal","title":"RDSNet: A New Deep Architecture for Reciprocal Object Detection and Instance Segmentation","date":"2019-12-11","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/spinenet-learning-scale-permuted-backbone-for","title":"SpineNet: Learning Scale-Permuted Backbone for Recognition and Localization","date":"2019-12-10","rows_on_this_dataset":10,"code_links":13,"syntology":null},{"paper":"/paper/solo-segmenting-objects-by-locations","title":"SOLO: Segmenting Objects by Locations","date":"2019-12-10","rows_on_this_dataset":2,"code_links":24,"syntology":null},{"paper":"/paper/bridging-the-gap-between-anchor-based-and","title":"Bridging the Gap Between Anchor-based and Anchor-free Detection via Adaptive Training Sample Selection","date":"2019-12-05","rows_on_this_dataset":1,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multiple-anchor-learning-for-visual-object","title":"Multiple Anchor Learning for Visual Object Detection","date":"2019-12-04","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":1,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/embedmask-embedding-coupling-for-one-stage","title":"EmbedMask: Embedding Coupling for One-stage Instance Segmentation","date":"2019-12-04","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/weakly-supervised-instance-segmentation-using-3","title":"Weakly Supervised Instance Segmentation using the Bounding Box Tightness Prior","date":"2019-12-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/one-shot-object-detection-with-co-attention-1","title":"One-Shot Object Detection with Co-Attention and Co-Excitation","date":"2019-11-28","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/soft-anchor-point-object-detection","title":"Soft Anchor-Point Object Detection","date":"2019-11-27","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/simple-pose-rethinking-and-improving-a-bottom","title":"Simple Pose: Rethinking and Improving a Bottom-up Approach for Multi-Person Pose Estimation","date":"2019-11-24","rows_on_this_dataset":3,"code_links":8,"syntology":null},{"paper":"/paper/panoptic-deeplab-a-simple-strong-and-fast","title":"Panoptic-DeepLab: A Simple, Strong, and Fast Baseline for Bottom-Up Panoptic Segmentation","date":"2019-11-22","rows_on_this_dataset":1,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":3,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multi-label-classification-with-label-graph","title":"Multi-Label Classification with Label Graph Superimposing","date":"2019-11-21","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-spatial-fusion-for-single-shot","title":"Learning Spatial Fusion for Single-Shot Object Detection","date":"2019-11-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/efficientdet-scalable-and-efficient-object","title":"EfficientDet: Scalable and Efficient Object Detection","date":"2019-11-20","rows_on_this_dataset":3,"code_links":64,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":70,"samples_ran":11,"samples_unverified":59,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/the-devil-is-in-the-details-delving-into","title":"The Devil is in the Details: Delving into Unbiased Data Processing for Human Pose Estimation","date":"2019-11-18","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/sognet-scene-overlap-graph-network-for","title":"SOGNet: Scene Overlap Graph Network for Panoptic Segmentation","date":"2019-11-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/directpose-direct-end-to-end-multi-person","title":"DirectPose: Direct End-to-End Multi-Person Pose Estimation","date":"2019-11-18","rows_on_this_dataset":2,"code_links":9,"syntology":null},{"paper":"/paper/centermask-real-time-anchor-free-instance-1","title":"CenterMask : Real-Time Anchor-Free Instance Segmentation","date":"2019-11-15","rows_on_this_dataset":16,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/191013321","title":"Semantic Object Accuracy for Generative Text-to-Image Synthesis","date":"2019-10-29","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":23,"samples_ran":4,"samples_unverified":19,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/spatialflow-bridging-all-tasks-for-panoptic","title":"SpatialFlow: Bridging All Tasks for Panoptic Segmentation","date":"2019-10-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/distribution-aware-coordinate-representation","title":"Distribution-Aware Coordinate Representation for Human Pose Estimation","date":"2019-10-14","rows_on_this_dataset":3,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":5,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deformable-kernels-adapting-effective","title":"Deformable Kernels: Adapting Effective Receptive Fields for Object Deformation","date":"2019-10-07","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/meta-r-cnn-towards-general-solver-for-1","title":"Meta R-CNN: Towards General Solver for Instance-Level Low-Shot Learning","date":"2019-10-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/meta-learning-to-detect-rare-objects","title":"Meta-Learning to Detect Rare Objects","date":"2019-10-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/hierarchical-shot-detector","title":"Hierarchical Shot Detector","date":"2019-10-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/polarmask-single-shot-instance-segmentation","title":"PolarMask: Single Shot Instance Segmentation with Polar Representation","date":"2019-09-29","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/190909777","title":"Generating Positive Bounding Boxes for Balanced Training of Object Detectors","date":"2019-09-21","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/adaptis-adaptive-instance-selection-network","title":"AdaptIS: Adaptive Instance Selection Network","date":"2019-09-17","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/pose-neural-fabrics-search","title":"Pose Neural Fabrics Search","date":"2019-09-16","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/cascade-rpn-delving-into-high-quality-region","title":"Cascade RPN: Delving into High-Quality Region Proposal Network with Adaptive Convolution","date":"2019-09-15","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/cbnet-a-novel-composite-backbone-network","title":"CBNet: A Novel Composite Backbone Network Architecture for Object Detection","date":"2019-09-09","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/visual-semantic-reasoning-for-image-text","title":"Visual Semantic Reasoning for Image-Text Matching","date":"2019-09-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/freeanchor-learning-to-match-anchors-for","title":"FreeAnchor: Learning to Match Anchors for Visual Object Detection","date":"2019-09-05","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/reflective-decoding-network-for-image","title":"Reflective Decoding Network for Image Captioning","date":"2019-08-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/bottom-up-higher-resolution-networks-for","title":"HigherHRNet: Scale-Aware Representation Learning for Bottom-Up Human Pose Estimation","date":"2019-08-27","rows_on_this_dataset":1,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":27,"samples_ran":5,"samples_unverified":22,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/single-stage-multi-person-pose-machines","title":"Single-Stage Multi-Person Pose Machines","date":"2019-08-24","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/generator-evaluator-selector-net-a-modular","title":"Generator evaluator-selector net for panoptic image segmentation and splitting unfamiliar objects into parts","date":"2019-08-24","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/instaboost-boosting-instance-segmentation-via","title":"InstaBoost: Boosting Instance Segmentation via Probability Map Guided Copy-Pasting","date":"2019-08-21","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/190807919","title":"Deep High-Resolution Representation Learning for Visual Recognition","date":"2019-08-20","rows_on_this_dataset":19,"code_links":42,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":34,"samples_ran":3,"samples_unverified":31,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unicoder-vl-a-universal-encoder-for-vision","title":"Unicoder-VL: A Universal Encoder for Vision and Language by Cross-modal Pre-training","date":"2019-08-16","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/matrix-nets-a-new-deep-architecture-for","title":"Matrix Nets: A New Deep Architecture for Object Detection","date":"2019-08-13","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/lip-local-importance-based-pooling","title":"LIP: Local Importance-based Pooling","date":"2019-08-12","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/few-shot-object-detection-with-attention-rpn","title":"Few-Shot Object Detection with Attention-RPN and Multi-Relation Detector","date":"2019-08-06","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":2,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attentive-normalization","title":"Attentive Normalization","date":"2019-08-04","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/compact-global-descriptor-for-neural-networks","title":"Compact Global Descriptor for Neural Networks","date":"2019-07-23","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/benchmarking-robustness-in-object-detection","title":"Benchmarking Robustness in Object Detection: Autonomous Driving when Winter is Coming","date":"2019-07-17","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-data-augmentation-strategies-for","title":"Learning Data Augmentation Strategies for Object Detection","date":"2019-06-26","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cascade-r-cnn-high-quality-object-detection","title":"Cascade R-CNN: High Quality Object Detection and Instance Segmentation","date":"2019-06-24","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/learning-instance-occlusion-for-panoptic","title":"Learning Instance Occlusion for Panoptic Segmentation","date":"2019-06-13","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/polysemous-visual-semantic-embedding-for-1","title":"Polysemous Visual-Semantic Embedding for Cross-Modal Retrieval","date":"2019-06-11","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/nas-fcos-fast-neural-architecture-search-for","title":"NAS-FCOS: Fast Neural Architecture Search for Object Detection","date":"2019-06-11","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rethinking-planar-homography-estimation-using","title":"Rethinking Planar Homography Estimation Using Perspective Fields","date":"2019-05-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cutmix-regularization-strategy-to-train","title":"CutMix: Regularization Strategy to Train Strong Classifiers with Localizable Features","date":"2019-05-13","rows_on_this_dataset":1,"code_links":30,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":24,"samples_ran":17,"samples_unverified":7,"pointer_only_for_licence":5,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/segmentation-is-all-you-need","title":"Segmentation is All You Need","date":"2019-04-30","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/reppoints-point-set-representation-for-object","title":"RepPoints: Point Set Representation for Object Detection","date":"2019-04-25","rows_on_this_dataset":10,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/gcnet-non-local-networks-meet-squeeze","title":"GCNet: Non-local Networks Meet Squeeze-Excitation Networks and Beyond","date":"2019-04-25","rows_on_this_dataset":5,"code_links":9,"syntology":null},{"paper":"/paper/an-energy-and-gpu-computation-efficient","title":"An Energy and GPU-Computation Efficient Backbone Network for Real-Time Object Detection","date":"2019-04-22","rows_on_this_dataset":2,"code_links":12,"syntology":null},{"paper":"/paper/190409925","title":"Attention Augmented Convolutional Networks","date":"2019-04-22","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":3,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/190408900","title":"CornerNet-Lite: Efficient Keypoint Based Object Detection","date":"2019-04-18","rows_on_this_dataset":3,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":27,"samples_ran":1,"samples_unverified":26,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/centernet-object-detection-with-keypoint","title":"CenterNet: Keypoint Triplets for Object Detection","date":"2019-04-17","rows_on_this_dataset":2,"code_links":20,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":2,"samples_unverified":9,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/objects-as-points","title":"Objects as Points","date":"2019-04-16","rows_on_this_dataset":1,"code_links":76,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":130,"samples_ran":10,"samples_unverified":120,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/foveabox-beyond-anchor-based-object-detector","title":"FoveaBox: Beyond Anchor-based Object Detector","date":"2019-04-08","rows_on_this_dataset":7,"code_links":7,"syntology":null},{"paper":"/paper/adaptively-connected-neural-networks","title":"Adaptively Connected Neural Networks","date":"2019-04-07","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/yolact-real-time-instance-segmentation","title":"YOLACT: Real-time Instance Segmentation","date":"2019-04-04","rows_on_this_dataset":2,"code_links":48,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":10,"samples_unverified":11,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/libra-r-cnn-towards-balanced-learning-for","title":"Libra R-CNN: Towards Balanced Learning for Object Detection","date":"2019-04-04","rows_on_this_dataset":2,"code_links":6,"syntology":null},{"paper":"/paper/res2net-a-new-multi-scale-backbone","title":"Res2Net: A New Multi-scale Backbone Architecture","date":"2019-04-02","rows_on_this_dataset":4,"code_links":34,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":3,"samples_unverified":6,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fcos-fully-convolutional-one-stage-object","title":"FCOS: Fully Convolutional One-Stage Object Detection","date":"2019-04-02","rows_on_this_dataset":5,"code_links":87,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":40,"samples_ran":13,"samples_unverified":27,"pointer_only_for_licence":18,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dm-gan-dynamic-memory-generative-adversarial","title":"DM-GAN: Dynamic Memory Generative Adversarial Networks for Text-to-Image Synthesis","date":"2019-04-02","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":4,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/tensormask-a-foundation-for-dense-object","title":"TensorMask: A Foundation for Dense Object Segmentation","date":"2019-03-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/feature-intertwiner-for-object-detection-1","title":"Feature Intertwiner for Object Detection","date":"2019-03-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/weight-standardization","title":"Micro-Batch Training with Batch-Channel Normalization and Weight Standardization","date":"2019-03-25","rows_on_this_dataset":2,"code_links":7,"syntology":null},{"paper":"/paper/pifpaf-composite-fields-for-human-pose","title":"PifPaf: Composite Fields for Human Pose Estimation","date":"2019-03-15","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/object-counting-and-instance-segmentation","title":"Object Counting and Instance Segmentation with Image-level Supervision","date":"2019-03-06","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/feature-selective-anchor-free-module-for","title":"Feature Selective Anchor-Free Module for Single-Shot Object Detection","date":"2019-03-02","rows_on_this_dataset":6,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/mask-scoring-r-cnn","title":"Mask Scoring R-CNN","date":"2019-03-01","rows_on_this_dataset":4,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deep-high-resolution-representation-learning","title":"Deep High-Resolution Representation Learning for Human Pose Estimation","date":"2019-02-25","rows_on_this_dataset":6,"code_links":39,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":25,"samples_ran":8,"samples_unverified":17,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/bottom-up-object-detection-by-grouping","title":"Bottom-up Object Detection by Grouping Extreme and Center Points","date":"2019-01-23","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":1,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/hybrid-task-cascade-for-instance-segmentation","title":"Hybrid Task Cascade for Instance Segmentation","date":"2019-01-22","rows_on_this_dataset":5,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":0,"samples_unverified":11,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/upsnet-a-unified-panoptic-segmentation","title":"UPSNet: A Unified Panoptic Segmentation Network","date":"2019-01-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/retinamask-learning-to-predict-masks-improves","title":"RetinaMask: Learning to predict masks improves state-of-the-art single-shot detection for free","date":"2019-01-10","rows_on_this_dataset":3,"code_links":53,"syntology":null},{"paper":"/paper/region-proposal-by-guided-anchoring","title":"Region Proposal by Guided Anchoring","date":"2019-01-10","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/panoptic-feature-pyramid-networks","title":"Panoptic Feature Pyramid Networks","date":"2019-01-08","rows_on_this_dataset":1,"code_links":12,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":7,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scale-aware-trident-networks-for-object","title":"Scale-Aware Trident Networks for Object Detection","date":"2019-01-07","rows_on_this_dataset":3,"code_links":4,"syntology":null},{"paper":"/paper/generating-multiple-objects-at-spatially","title":"Generating Multiple Objects at Spatially Distinct Locations","date":"2019-01-03","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/rethinking-on-multi-stage-networks-for-human","title":"Rethinking on Multi-Stage Networks for Human Pose Estimation","date":"2019-01-01","rows_on_this_dataset":5,"code_links":7,"syntology":null},{"paper":"/paper/openpose-realtime-multi-person-2d-pose","title":"OpenPose: Realtime Multi-Person 2D Pose Estimation using Part Affinity Fields","date":"2018-12-18","rows_on_this_dataset":1,"code_links":51,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":3,"samples_unverified":13,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/posefix-model-agnostic-general-human-pose","title":"PoseFix: Model-agnostic General Human Pose Refinement Network","date":"2018-12-10","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/attention-guided-unified-network-for-panoptic","title":"Attention-guided Unified Network for Panoptic Segmentation","date":"2018-12-10","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/few-shot-object-detection-via-feature","title":"Few-shot Object Detection via Feature Reweighting","date":"2018-12-05","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-to-fuse-things-and-stuff","title":"Learning to Fuse Things and Stuff","date":"2018-12-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/grid-r-cnn","title":"Grid R-CNN","date":"2018-11-29","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":0,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/one-shot-instance-segmentation","title":"One-Shot Instance Segmentation","date":"2018-11-28","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":0,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deformable-convnets-v2-more-deformable-better","title":"Deformable ConvNets v2: More Deformable, Better Results","date":"2018-11-27","rows_on_this_dataset":3,"code_links":26,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":2,"samples_unverified":11,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/polarity-loss-for-zero-shot-object-detection","title":"Polarity Loss for Zero-shot Object Detection","date":"2018-11-22","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/rethinking-imagenet-pre-training","title":"Rethinking ImageNet Pre-training","date":"2018-11-21","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/gradient-harmonized-single-stage-detector","title":"Gradient Harmonized Single-stage Detector","date":"2018-11-13","rows_on_this_dataset":2,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/m2det-a-single-shot-object-detector-based-on","title":"M2Det: A Single-Shot Object Detector based on Multi-Level Feature Pyramid Network","date":"2018-11-12","rows_on_this_dataset":6,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/softer-nms-rethinking-bounding-box-regression","title":"Bounding Box Regression with Uncertainty for Accurate Object Detection","date":"2018-09-23","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/panoptic-segmentation-with-a-joint-semantic","title":"Panoptic Segmentation with a Joint Semantic and Instance Segmentation Network","date":"2018-09-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/associating-inter-image-salient-instances-for","title":"Associating Inter-Image Salient Instances for Weakly Supervised Semantic Segmentation","date":"2018-09-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multimodal-differential-network-for-visual","title":"Multimodal Differential Network for Visual Question Generation","date":"2018-08-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/question-guided-hybrid-convolution-for-visual","title":"Question-Guided Hybrid Convolution for Visual Question Answering","date":"2018-08-08","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cornernet-detecting-objects-as-paired","title":"CornerNet: Detecting Objects as Paired Keypoints","date":"2018-08-03","rows_on_this_dataset":3,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/acquisition-of-localization-confidence-for","title":"Acquisition of Localization Confidence for Accurate Object Detection","date":"2018-07-30","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":2,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/where-are-the-blobs-counting-by-localization","title":"Where are the Blobs: Counting by Localization with Point Supervision","date":"2018-07-25","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/invariant-information-distillation-for","title":"Invariant Information Clustering for Unsupervised Image Classification and Segmentation","date":"2018-07-17","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":18,"samples_ran":2,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/multiposenet-fast-multi-person-pose","title":"MultiPoseNet: Fast Multi-Person Pose Estimation using Pose Residual Network","date":"2018-07-11","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/r-vqa-learning-visual-relation-facts-with","title":"R-VQA: Learning Visual Relation Facts with Semantic Attention for Visual Question Answering","date":"2018-05-24","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/sniper-efficient-multi-scale-training","title":"SNIPER: Efficient Multi-Scale Training","date":"2018-05-23","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/simple-baselines-for-human-pose-estimation","title":"Simple Baselines for Human Pose Estimation and Tracking","date":"2018-04-17","rows_on_this_dataset":5,"code_links":27,"syntology":null},{"paper":"/paper/yolov3-an-incremental-improvement","title":"YOLOv3: An Incremental Improvement","date":"2018-04-08","rows_on_this_dataset":1,"code_links":311,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":124,"samples_ran":18,"samples_unverified":106,"pointer_only_for_licence":19,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/personlab-person-pose-estimation-and-instance","title":"PersonLab: Person Pose Estimation and Instance Segmentation with a Bottom-Up, Part-Based, Geometric Embedding Model","date":"2018-03-22","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/group-normalization","title":"Group Normalization","date":"2018-03-22","rows_on_this_dataset":3,"code_links":22,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":5,"samples_unverified":10,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/stacked-cross-attention-for-image-text","title":"Stacked Cross Attention for Image-Text Matching","date":"2018-03-21","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":7,"samples_unverified":9,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/path-aggregation-network-for-instance","title":"Path Aggregation Network for Instance Segmentation","date":"2018-03-05","rows_on_this_dataset":3,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/lstd-a-low-shot-transfer-detector-for-object","title":"LSTD: A Low-Shot Transfer Detector for Object Detection","date":"2018-03-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/chatpainter-improving-text-to-image","title":"ChatPainter: Improving Text to Image Generation using Dialogue","date":"2018-02-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/pose-flow-efficient-online-pose-tracking","title":"Pose Flow: Efficient Online Pose Tracking","date":"2018-02-03","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/detect-and-track-efficient-pose-estimation-in","title":"Detect-and-Track: Efficient Pose Estimation in Videos","date":"2017-12-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/masklab-instance-segmentation-by-refining","title":"MaskLab: Instance Segmentation by Refining Object Detection with Semantic and Direction Features","date":"2017-12-13","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-semantic-concepts-and-order-for","title":"Learning Semantic Concepts and Order for Image and Sentence Matching","date":"2017-12-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cascade-r-cnn-delving-into-high-quality","title":"Cascade R-CNN: Delving into High Quality Object Detection","date":"2017-12-03","rows_on_this_dataset":6,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/attngan-fine-grained-text-to-image-generation","title":"AttnGAN: Fine-Grained Text to Image Generation with Attentional Generative Adversarial Networks","date":"2017-11-28","rows_on_this_dataset":1,"code_links":20,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":3,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/weakly-supervised-object-discovery-by","title":"Weakly Supervised Object Discovery by Generative Adversarial & Ranking Networks","date":"2017-11-22","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/an-analysis-of-scale-invariance-in-object-1","title":"An Analysis of Scale Invariance in Object Detection - SNIP","date":"2017-11-22","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/non-local-neural-networks","title":"Non-local Neural Networks","date":"2017-11-21","rows_on_this_dataset":7,"code_links":32,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cascaded-pyramid-network-for-multi-person","title":"Cascaded Pyramid Network for Multi-Person Pose Estimation","date":"2017-11-20","rows_on_this_dataset":7,"code_links":5,"syntology":null},{"paper":"/paper/single-shot-refinement-neural-network-for","title":"Single-Shot Refinement Neural Network for Object Detection","date":"2017-11-18","rows_on_this_dataset":3,"code_links":16,"syntology":null},{"paper":"/paper/co-attending-free-form-regions-and-detections","title":"Co-attending Free-form Regions and Detections with Multi-modal Multiplicative Feature Embedding for Visual Question Answering","date":"2017-11-18","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/dual-path-convolutional-image-text-embedding","title":"Dual-Path Convolutional Image-Text Embeddings with Instance Loss","date":"2017-11-15","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/high-order-attention-models-for-visual","title":"High-Order Attention Models for Visual Question Answering","date":"2017-11-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/stackgan-realistic-image-synthesis-with","title":"StackGAN++: Realistic Image Synthesis with Stacked Generative Adversarial Networks","date":"2017-10-19","rows_on_this_dataset":1,"code_links":16,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":31,"samples_ran":10,"samples_unverified":21,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/soft-proposal-networks-for-weakly-supervised","title":"Soft Proposal Networks for Weakly Supervised Object Localization","date":"2017-09-06","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/chainercv-a-library-for-deep-learning-in","title":"ChainerCV: a Library for Deep Learning in Computer Vision","date":"2017-08-28","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/focal-loss-for-dense-object-detection","title":"Focal Loss for Dense Object Detection","date":"2017-08-07","rows_on_this_dataset":2,"code_links":234,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":11,"samples_unverified":0,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unsupervised-object-discovery-and-co","title":"Unsupervised Object Discovery and Co-Localization by Deep Descriptor Transforming","date":"2017-07-20","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/revisiting-unreasonable-effectiveness-of-data","title":"Revisiting Unreasonable Effectiveness of Data in Deep Learning Era","date":"2017-07-10","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/wildcat-weakly-supervised-learning-of-deep","title":"WILDCAT: Weakly Supervised Learning of Deep ConvNets for Image Classification, Pointwise Localization and Segmentation","date":"2017-07-01","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/few-example-object-detection-with-model","title":"Few-Example Object Detection with Model Communication","date":"2017-06-26","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mask-r-cnn","title":"Mask R-CNN","date":"2017-03-20","rows_on_this_dataset":9,"code_links":179,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":140,"samples_ran":42,"samples_unverified":98,"pointer_only_for_licence":23,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deformable-convolutional-networks","title":"Deformable Convolutional Networks","date":"2017-03-17","rows_on_this_dataset":1,"code_links":38,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":3,"samples_unverified":6,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/learning-spatial-regularization-with-image","title":"Learning Spatial Regularization with Image-level Supervisions for Multi-label Image Classification","date":"2017-02-20","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/towards-accurate-multi-person-pose-estimation","title":"Towards Accurate Multi-person Pose Estimation in the Wild","date":"2017-01-06","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/beyond-skip-connections-top-down-modulation","title":"Beyond Skip Connections: Top-Down Modulation for Object Detection","date":"2016-12-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/feature-pyramid-networks-for-object-detection","title":"Feature Pyramid Networks for Object Detection","date":"2016-12-09","rows_on_this_dataset":2,"code_links":85,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":51,"samples_ran":16,"samples_unverified":35,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/making-the-v-in-vqa-matter-elevating-the-role","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","date":"2016-12-02","rows_on_this_dataset":2,"code_links":7,"syntology":null},{"paper":"/paper/rmpe-regional-multi-person-pose-estimation","title":"RMPE: Regional Multi-person Pose Estimation","date":"2016-12-01","rows_on_this_dataset":5,"code_links":14,"syntology":null},{"paper":"/paper/weakly-supervised-cascaded-convolutional","title":"Weakly Supervised Cascaded Convolutional Networks","date":"2016-11-24","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/realtime-multi-person-2d-pose-estimation","title":"Realtime Multi-Person 2D Pose Estimation using Part Affinity Fields","date":"2016-11-24","rows_on_this_dataset":3,"code_links":61,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":23,"samples_ran":4,"samples_unverified":19,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fully-convolutional-instance-aware-semantic","title":"Fully Convolutional Instance-aware Semantic Segmentation","date":"2016-11-23","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":0,"samples_unverified":5,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/associative-embedding-end-to-end-learning-for","title":"Associative Embedding: End-to-End Learning for Joint Detection and Grouping","date":"2016-11-16","rows_on_this_dataset":3,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":0,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/graph-structured-representations-for-visual","title":"Graph-Structured Representations for Visual Question Answering","date":"2016-09-19","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/training-recurrent-answering-units-with-joint","title":"Training Recurrent Answering Units with Joint Loss Minimization for VQA","date":"2016-06-12","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/multimodal-compact-bilinear-pooling-for","title":"Multimodal Compact Bilinear Pooling for Visual Question Answering and Visual Grounding","date":"2016-06-06","rows_on_this_dataset":2,"code_links":10,"syntology":null},{"paper":"/paper/multimodal-residual-learning-for-visual-qa","title":"Multimodal Residual Learning for Visual QA","date":"2016-06-05","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hierarchical-question-image-co-attention-for","title":"Hierarchical Question-Image Co-Attention for Visual Question Answering","date":"2016-05-31","rows_on_this_dataset":2,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":1,"samples_unverified":6,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/counting-everyday-objects-in-everyday-scenes","title":"Counting Everyday Objects in Everyday Scenes","date":"2016-04-12","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/a-multipath-network-for-object-detection","title":"A MultiPath Network for Object Detection","date":"2016-04-07","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-focused-dynamic-attention-model-for-visual","title":"A Focused Dynamic Attention Model for Visual Question Answering","date":"2016-04-06","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/image-captioning-and-visual-question","title":"Image Captioning and Visual Question Answering Based on Attributes and External Knowledge","date":"2016-03-09","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dynamic-memory-networks-for-visual-and","title":"Dynamic Memory Networks for Visual and Textual Question Answering","date":"2016-03-04","rows_on_this_dataset":1,"code_links":10,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":7,"samples_unverified":0,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/weakly-supervised-localization-using-deep","title":"Weakly Supervised Localization using Deep Feature Maps","date":"2016-03-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/instance-aware-semantic-segmentation-via","title":"Instance-aware Semantic Segmentation via Multi-task Network Cascades","date":"2015-12-14","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/neural-self-talk-image-understanding-via","title":"Neural Self Talk: Image Understanding via Continuous Questioning and Answering","date":"2015-12-10","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/deep-residual-learning-for-image-recognition","title":"Deep Residual Learning for Image Recognition","date":"2015-12-10","rows_on_this_dataset":3,"code_links":484,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":377,"samples_ran":230,"samples_unverified":147,"pointer_only_for_licence":187,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/simple-baseline-for-visual-question-answering","title":"Simple Baseline for Visual Question Answering","date":"2015-12-07","rows_on_this_dataset":2,"code_links":7,"syntology":null},{"paper":"/paper/ask-attend-and-answer-exploring-question","title":"Ask, Attend and Answer: Exploring Question-Guided Spatial Attention for Visual Question Answering","date":"2015-11-17","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pronet-learning-to-propose-object-specific","title":"ProNet: Learning to Propose Object-specific Boxes for Cascaded Neural Networks","date":"2015-11-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/weakly-supervised-deep-detection-networks","title":"Weakly Supervised Deep Detection Networks","date":"2015-11-09","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/stacked-attention-networks-for-image-question","title":"Stacked Attention Networks for Image Question Answering","date":"2015-11-07","rows_on_this_dataset":1,"code_links":16,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vqa-visual-question-answering","title":"VQA: Visual Question Answering","date":"2015-05-03","rows_on_this_dataset":10,"code_links":21,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":6,"samples_unverified":1,"pointer_only_for_licence":6,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/deep-visual-semantic-alignments-for","title":"Deep Visual-Semantic Alignments for Generating Image Descriptions","date":"2014-12-07","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":261,"samples_harvested":4106,"samples_ran":1761,"samples_unverified":2345,"pointer_only_for_licence":923,"papers_with_no_sample_that_ran":36,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}