{"url":"/dataset/flickr30k-cna","name":"Flickr30k-CNA","full_name":"Flickr30k-Chinese All","description_markdown":"Former Flickr30k-CN translates the training and validation sets of Flickr30k using machine translation and manually translates the test set. We check the machine-translated results and find two kinds of problems. (1) Some sentences have language problems and translation errors. (2) Some sentences have poor semantics. In addition, the different translation ways between the training set and test set prevent the model from achieving accurate performance. We gather 6 professional English and Chinese linguists to meticulously re-translate all data of Flickr30k and double-check each sentence.","description_withheld":null,"homepage":"https://zero.so.com","introduced_date":"2022-05-08","introduced_date_note":null,"introduced_by":{"paper":"/paper/zero-and-r2d2-a-large-scale-chinese-cross","title":"CCMB: A Large-scale Chinese Cross-modal Benchmark","first_author":"Chunyu Xie","url":null},"license":null,"modalities":[],"tasks":[{"name":"Image Retrieval","url":"/task/image-retrieval","datasets_with_task":"/datasets/task/image-retrieval"},{"name":"Zero-shot Image Retrieval","url":"/task/zero-shot-image-retrieval","datasets_with_task":"/datasets/task/zero-shot-image-retrieval"},{"name":"Zero-shot Text Retrieval","url":"/task/zero-shot-text-retrieval","datasets_with_task":"/datasets/task/zero-shot-text-retrieval"}],"languages":[],"variants":["Flickr30k-CNA","Flickr30k-CN"],"data_loaders":[],"num_papers_in_archive":7,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/image-retrieval-on-flickr30k-cn","task":"Image Retrieval","dataset_variant":"Flickr30k-CN","rows":11,"metrics":["R@1","R@10","R@5"],"first_row_in_archive_order":{"model":"InternVL-G-FT","paper":"/paper/internvl-scaling-up-vision-foundation-models","metrics":{"R@1":"85.9","R@10":"97.1","R@5":"98.7"},"code_links":[{"title":"opengvlab/internvl","url":"https://github.com/opengvlab/internvl"},{"title":"opengvlab/internvl-mmdetseg","url":"https://github.com/opengvlab/internvl-mmdetseg"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/internvl-scaling-up-vision-foundation-models","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","date":"2023-12-21","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":2,"samples_unverified":0,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/chinese-clip-contrastive-vision-language","title":"Chinese CLIP: Contrastive Vision-Language Pretraining in Chinese","date":"2022-11-02","rows_on_this_dataset":5,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/zero-and-r2d2-a-large-scale-chinese-cross","title":"CCMB: A Large-scale Chinese Cross-modal Benchmark","date":"2022-05-08","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":9,"samples_ran":0,"samples_unverified":9,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/wukong-100-million-large-scale-chinese-cross","title":"Wukong: A 100 Million Large-scale Chinese Cross-modal Pre-training Benchmark","date":"2022-02-14","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":20,"samples_ran":6,"samples_unverified":14,"pointer_only_for_licence":2,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}