{"url":"/dataset/cuhk-pedes","name":"CUHK-PEDES","full_name":"CUHK-PEDES","description_markdown":"The **CUHK-PEDES** dataset is a caption-annotated pedestrian dataset. It contains 40,206 images over 13,003 persons. Images are collected from five existing person re-identification datasets, CUHK03, Market-1501, SSM, VIPER, and CUHK01 while each image is annotated with 2 text descriptions by crowd-sourcing workers. Sentences incorporate rich details about person appearances, actions, poses.\r\n\r\nSource: [MGD-GAN: Text-to-Pedestrian generation through Multi-Grained Discrimination](https://arxiv.org/abs/2010.00947)\r\nImage Source: [https://www.researchgate.net/figure/Image-samples-in-three-datasets-For-MSCOCO-and-Flickr30k-dataset-we-view-every-image_fig2_321095980](https://www.researchgate.net/figure/Image-samples-in-three-datasets-For-MSCOCO-and-Flickr30k-dataset-we-view-every-image_fig2_321095980)","description_withheld":null,"homepage":"https://github.com/layumi/Image-Text-Embedding/tree/master/dataset/CUHK-PEDES-prepare","introduced_date":"2017-02-19","introduced_date_note":null,"introduced_by":{"paper":"/paper/person-search-with-natural-language","title":"Person Search with Natural Language Description","first_author":"Shuang Li","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Cross-Modal Retrieval","url":"/task/cross-modal-retrieval","datasets_with_task":"/datasets/task/cross-modal-retrieval"},{"name":"Text based Person Retrieval","url":"/task/nlp-based-person-retrival","datasets_with_task":"/datasets/task/nlp-based-person-retrival"},{"name":"Text-based Person Retrieval with Noisy Correspondence","url":"/task/text-based-person-retrieval-with-noisy","datasets_with_task":"/datasets/task/text-based-person-retrieval-with-noisy"},{"name":"Pedestrian Image Caption","url":"/task/pedestrian-image-caption","datasets_with_task":"/datasets/task/pedestrian-image-caption"}],"languages":[],"variants":["CUHK-PEDES"],"data_loaders":[{"repo":"https://github.com/layumi/Image-Text-Embedding","url":"https://github.com/layumi/Image-Text-Embedding","frameworks":[]}],"num_papers_in_archive":93,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/nlp-based-person-retrival-on-cuhk-pedes","task":"Text based Person Retrieval","dataset_variant":"CUHK-PEDES","rows":21,"metrics":["R@1","R@5","R@10","mAP","Rank-1","Rank-10","Rank-5"],"first_row_in_archive_order":{"model":"MARS","paper":"/paper/mars-paying-more-attention-to-visual","metrics":{"R@1":"77.62","R@10":"94.27","R@5":"90.63","mAP":"71.41"},"code_links":[{"title":"ergastialex/mars","url":"https://github.com/ergastialex/mars"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-based-person-retrieval-with-noisy","task":"Text-based Person Retrieval with Noisy Correspondence","dataset_variant":"CUHK-PEDES","rows":6,"metrics":["Rank-1","Rank 10","Rank-5","mAP","mINP"],"first_row_in_archive_order":{"model":"RDE","paper":"/paper/noisy-correspondence-learning-for-text-to","metrics":{"Rank 10":"93.63","Rank-1":"74.46","Rank-5":"89.42","mAP":"66.13","mINP":"49.66"},"code_links":[{"title":"QinYang79/RDE","url":"https://github.com/QinYang79/RDE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/cross-modal-retrieval-on-cuhk-pedes","task":"Cross-Modal Retrieval","dataset_variant":"CUHK-PEDES","rows":1,"metrics":["Text-to-image Medr"],"first_row_in_archive_order":{"model":"Dual Path","paper":"/paper/dual-path-convolutional-image-text-embedding","metrics":{"Text-to-image Medr":"2"},"code_links":[{"title":"layumi/Image-Text-Embedding","url":"https://github.com/layumi/Image-Text-Embedding"},{"title":"pshroff04/Dual_Path_CNN","url":"https://github.com/pshroff04/Dual_Path_CNN"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/mars-paying-more-attention-to-visual","title":"MARS: Paying more attention to visual attributes for text-based person search","date":"2024-07-05","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/from-data-deluge-to-data-curation-a-filtering","title":"From Data Deluge to Data Curation: A Filtering-WoRA Paradigm for Efficient Text-based Person Search","date":"2024-04-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/cross-modal-adaptive-dual-association-for","title":"Cross-Modal Adaptive Dual Association for Text-to-Image Person Retrieval","date":"2023-12-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/noisy-correspondence-learning-for-text-to","title":"Noisy-Correspondence Learning for Text-to-Image Person Re-identification","date":"2023-08-19","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":12,"samples_unverified":2,"pointer_only_for_licence":14,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-unified-text-based-person-retrieval-a","title":"Towards Unified Text-based Person Retrieval: A Large-scale Multi-Attribute and Language Search Benchmark","date":"2023-06-05","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":6,"samples_unverified":6,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rasa-relation-and-sensitivity-aware","title":"RaSa: Relation and Sensitivity Aware Representation Learning for Text-based Person Search","date":"2023-05-23","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":3,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/cross-modal-implicit-relation-reasoning-and","title":"Cross-Modal Implicit Relation Reasoning and Aligning for Text-to-Image Person Retrieval","date":"2023-03-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":5,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-simple-and-robust-correlation-filtering","title":"A Simple and Robust Correlation Filtering Method for Text-based Person Search","date":"2022-11-04","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/deep-evidential-learning-with-noisy","title":"Deep Evidential Learning with Noisy Correspondence for Cross-Modal Retrieval","date":"2022-10-10","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/see-finer-see-more-implicit-modality","title":"See Finer, See More: Implicit Modality Alignment for Text-based Person Retrieval","date":"2022-08-18","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-semantic-aligned-feature","title":"Learning Semantic-Aligned Feature Representation for Text-based Person Search","date":"2021-12-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/text-based-person-search-with-limited-data","title":"Text-Based Person Search with Limited Data","date":"2021-10-20","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dssl-deep-surroundings-person-separation","title":"DSSL: Deep Surroundings-person Separation Learning for Text-based Person Retrieval","date":"2021-09-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/semantically-self-aligned-network-for-text-to","title":"Semantically Self-Aligned Network for Text-to-Image Part-aware Person Re-identification","date":"2021-07-27","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/tipcb-a-simple-but-effective-part-based","title":"TIPCB: A Simple but Effective Part-based Convolutional Baseline for Text-based Person Search","date":"2021-05-25","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/learning-transferable-visual-models-from","title":"Learning Transferable Visual Models From Natural Language Supervision","date":"2021-02-26","rows_on_this_dataset":1,"code_links":82,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":20,"samples_ran":16,"samples_unverified":4,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/axm-net-cross-modal-context-sharing-attention","title":"AXM-Net: Implicit Cross-Modal Feature Alignment for Person Re-identification","date":"2021-01-19","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/contextual-non-local-alignment-over-full","title":"Contextual Non-Local Alignment over Full-Scale Representation for Text-Based Person Search","date":"2021-01-08","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/hierarchical-gumbel-attention-network-for","title":"Hierarchical Gumbel Attention Network for Text-based Person Search","date":"2020-10-10","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/vitaa-visual-textual-attributes-alignment-in","title":"ViTAA: Visual-Textual Attributes Alignment in Person Search by Natural Language","date":"2020-05-15","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/text-based-person-search-via-attribute-aided","title":"Text-based Person Search via Attribute-aided Matching","date":"2020-03-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/improving-description-based-person-re","title":"Improving Description-based Person Re-identification by Multi-granularity Image-text Alignments","date":"2019-06-23","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/deep-cross-modal-projection-learning-for","title":"Deep Cross-Modal Projection Learning for Image-Text Matching","date":"2018-09-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-deep-visual-representation-for","title":"Improving Deep Visual Representation for Person Re-identification by Global and Local Image-language Association","date":"2018-08-05","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dual-path-convolutional-image-text-embedding","title":"Dual-Path Convolutional Image-Text Embeddings with Instance Loss","date":"2017-11-15","rows_on_this_dataset":2,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":6,"samples_harvested":63,"samples_ran":43,"samples_unverified":20,"pointer_only_for_licence":30,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}