{"url":"/dataset/imagenet-sketch","name":"ImageNet-Sketch","full_name":null,"description_markdown":"ImageNet-Sketch data set consists of 50,889 images,  approximately 50 images for each of the 1000 ImageNet classes. The data set is constructed with Google Image queries \"sketch of __\", where __ is the standard class name. Only within the \"black and white\" color scheme is searched. 100 images are initially queried for every class, and the pulled images are cleaned by deleting the irrelevant images and images that are for similar but different classes. For some classes, there are less than 50 images after manually cleaning, and then the data set is augmented by flipping and rotating the images.\r\n\r\nSource: [ImageNet-Sketch](https://github.com/HaohanWang/ImageNet-Sketch)\r\nImage Source: [https://github.com/HaohanWang/ImageNet-Sketch](https://github.com/HaohanWang/ImageNet-Sketch)","description_withheld":null,"homepage":"https://github.com/HaohanWang/ImageNet-Sketch","introduced_date":"2019-05-29","introduced_date_note":null,"introduced_by":{"paper":"/paper/190513549","title":"Learning Robust Global Representations by Penalizing Local Predictive Power","first_author":"Haohan Wang","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Image Classification","url":"/task/image-classification","datasets_with_task":"/datasets/task/image-classification"},{"name":"Domain Adaptation","url":"/task/domain-adaptation","datasets_with_task":"/datasets/task/domain-adaptation"},{"name":"Domain Generalization","url":"/task/domain-generalization","datasets_with_task":"/datasets/task/domain-generalization"},{"name":"Zero-Shot Transfer Image Classification","url":"/task/zero-shot-transfer-image-classification","datasets_with_task":"/datasets/task/zero-shot-transfer-image-classification"},{"name":"Data Augmentation","url":"/task/data-augmentation","datasets_with_task":"/datasets/task/data-augmentation"}],"languages":[],"variants":["ImageNet-Sketch"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/songweig/imagenet_sketch","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/imagenet_sketch","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/imagenet_sketch","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/HaohanWang/ImageNet-Sketch","url":"https://github.com/HaohanWang/ImageNet-Sketch","frameworks":["pytorch"]}],"num_papers_in_archive":268,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/domain-generalization-on-imagenet-sketch","task":"Domain Generalization","dataset_variant":"ImageNet-Sketch","rows":20,"metrics":["Top-1 accuracy"],"first_row_in_archive_order":{"model":"Model soups (BASIC-L)","paper":"/paper/model-soups-averaging-weights-of-multiple","metrics":{"Top-1 accuracy":"77.18"},"code_links":[{"title":"mlfoundations/model-soups","url":"https://github.com/mlfoundations/model-soups"},{"title":"Burf/ModelSoups","url":"https://github.com/Burf/ModelSoups"},{"title":"facebookresearch/ModelRatatouille","url":"https://github.com/facebookresearch/ModelRatatouille"},{"title":"hwk0702/keras2torch","url":"https://github.com/hwk0702/keras2torch/tree/main/Computer_Vision/Model_Soup"},{"title":"flowritecom/flow-merge","url":"https://github.com/flowritecom/flow-merge"},{"title":"shallowlearn/sportsreid","url":"https://github.com/shallowlearn/sportsreid"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/zero-shot-transfer-image-classification-on-8","task":"Zero-Shot Transfer Image Classification","dataset_variant":"ImageNet-Sketch","rows":7,"metrics":["Accuracy (Private)"],"first_row_in_archive_order":{"model":"CoCa","paper":"/paper/coca-contrastive-captioners-are-image-text","metrics":{"Accuracy (Private)":"77.6"},"code_links":[{"title":"mlfoundations/open_clip","url":"https://github.com/mlfoundations/open_clip"},{"title":"facebookresearch/multimodal","url":"https://github.com/facebookresearch/multimodal"},{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"amitakamath/whatsup_vlms","url":"https://github.com/amitakamath/whatsup_vlms"},{"title":"amitakamath/hard_positives","url":"https://github.com/amitakamath/hard_positives"},{"title":"Chaolei98/FreeZAD","url":"https://github.com/Chaolei98/FreeZAD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-classification-on-imagenet-sketch","task":"Image Classification","dataset_variant":"ImageNet-Sketch","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"µ2Net+ (ViT-L/16)","paper":"/paper/a-continual-development-methodology-for-large","metrics":{"Accuracy":"88.6"},"code_links":[{"title":"google-research/google-research","url":"https://github.com/google-research/google-research/tree/master/muNet"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/eva-clip-18b-scaling-clip-to-18-billion","title":"EVA-CLIP-18B: Scaling CLIP to 18 Billion Parameters","date":"2024-02-06","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":1,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/internvl-scaling-up-vision-foundation-models","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","date":"2023-12-21","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/distilling-out-of-distribution-robustness-1","title":"Distilling Out-of-Distribution Robustness from Vision-Language Foundation Models","date":"2023-11-02","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/eva-clip-improved-training-techniques-for","title":"EVA-CLIP: Improved Training Techniques for CLIP at Scale","date":"2023-03-27","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-whac-a-mole-dilemma-shortcuts-come-in","title":"A Whac-A-Mole Dilemma: Shortcuts Come in Multiples Where Mitigating One Amplifies Others","date":"2022-12-09","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/context-aware-robust-fine-tuning","title":"Context-Aware Robust Fine-Tuning","date":"2022-11-29","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/altclip-altering-the-language-encoder-in-clip","title":"AltCLIP: Altering the Language Encoder in CLIP for Extended Language Capabilities","date":"2022-11-12","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":2,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/metaformer-baselines-for-vision","title":"MetaFormer Baselines for Vision","date":"2022-10-24","rows_on_this_dataset":6,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/generalized-parametric-contrastive-learning","title":"Generalized Parametric Contrastive Learning","date":"2022-09-26","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/enhance-the-visual-representation-via","title":"Enhance the Visual Representation via Discrete Adversarial Training","date":"2022-09-16","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":10,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-continual-development-methodology-for-large","title":"A Continual Development Methodology for Large-scale Multitask Dynamic ML Systems","date":"2022-09-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/sequencer-deep-lstm-for-image-classification","title":"Sequencer: Deep LSTM for Image Classification","date":"2022-05-04","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":4,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":9,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/model-soups-averaging-weights-of-multiple","title":"Model soups: averaging weights of multiple fine-tuned models improves accuracy without increasing inference time","date":"2022-03-10","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":5,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vision-models-are-more-robust-and-fair-when","title":"Vision Models Are More Robust And Fair When Pretrained On Uncurated Images Without Supervision","date":"2022-02-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-convnet-for-the-2020s","title":"A ConvNet for the 2020s","date":"2022-01-10","rows_on_this_dataset":1,"code_links":54,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":80,"samples_ran":54,"samples_unverified":26,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/pyramid-adversarial-training-improves-vit","title":"Pyramid Adversarial Training Improves ViT Performance","date":"2021-11-30","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/discrete-representations-strengthen-vision-1","title":"Discrete Representations Strengthen Vision Transformer Robustness","date":"2021-11-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/combined-scaling-for-zero-shot-transfer","title":"Combined Scaling for Zero-shot Transfer Learning","date":"2021-11-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/masked-autoencoders-are-scalable-vision","title":"Masked Autoencoders Are Scalable Vision Learners","date":"2021-11-11","rows_on_this_dataset":1,"code_links":58,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":137,"samples_ran":71,"samples_unverified":66,"pointer_only_for_licence":73,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":12,"samples_harvested":299,"samples_ran":159,"samples_unverified":140,"pointer_only_for_licence":87,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}