{"url":"/dataset/rsicd","name":"RSICD","full_name":"Remote Sensing Image Captioning Dataset","description_markdown":null,"description_withheld":null,"homepage":"https://github.com/201528014227051/RSICD_optimal","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/exploring-models-and-data-for-remote-sensing","title":"Exploring Models and Data for Remote Sensing Image Caption Generation","first_author":"Xiaoqiang Lu","url":null},"license":null,"modalities":[{"name":"Images","url":"/datasets/modality/images"}],"tasks":[{"name":"Image Captioning","url":"/task/image-captioning","datasets_with_task":"/datasets/task/image-captioning"},{"name":"Cross-Modal Retrieval","url":"/task/cross-modal-retrieval","datasets_with_task":"/datasets/task/cross-modal-retrieval"},{"name":"Text Retrieval","url":"/task/text-retrieval","datasets_with_task":"/datasets/task/text-retrieval"},{"name":"Image-to-Text Retrieval","url":"/task/image-to-text-retrieval","datasets_with_task":"/datasets/task/image-to-text-retrieval"},{"name":"Scene Classification","url":"/task/scene-classification","datasets_with_task":"/datasets/task/scene-classification"}],"languages":[],"variants":["RSICD"],"data_loaders":[{"repo":"https://github.com/isaaccorley/torchrs","url":"https://github.com/isaaccorley/torchrs","frameworks":["pytorch"]},{"repo":"https://github.com/201528014227051/RSICD_optimal","url":"https://github.com/201528014227051/RSICD_optimal","frameworks":[]}],"num_papers_in_archive":70,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/cross-modal-retrieval-on-rsicd","task":"Cross-Modal Retrieval","dataset_variant":"RSICD","rows":10,"metrics":["Mean Recall","Image-to-text R@1","text-to-image R@1"],"first_row_in_archive_order":{"model":"HarMA (w/ GeoRSCLIP)","paper":"/paper/efficient-remote-sensing-with-harmonized","metrics":{"Image-to-text R@1":"20.52%","Mean Recall":"38.95%","text-to-image R@1":"15.84%"},"code_links":[{"title":"seekerhuang/harma","url":"https://github.com/seekerhuang/harma"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/image-to-text-retrieval-on-rsicd","task":"Image-to-Text Retrieval","dataset_variant":"RSICD","rows":1,"metrics":["Image to Text Recall@1"],"first_row_in_archive_order":{"model":"GeoRSCLIP-FT","paper":"/paper/rs5m-a-large-scale-vision-language-dataset","metrics":{"Image to Text Recall@1":"22.14%"},"code_links":[{"title":"om-ai-lab/rs5m","url":"https://github.com/om-ai-lab/rs5m"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-retrieval-on-rsicd","task":"Text Retrieval","dataset_variant":"RSICD","rows":1,"metrics":["Recall@1"],"first_row_in_archive_order":{"model":"GeoRSCLIP-FT","paper":"/paper/rs5m-a-large-scale-vision-language-dataset","metrics":{"Recall@1":"15.59%"},"code_links":[{"title":"om-ai-lab/rs5m","url":"https://github.com/om-ai-lab/rs5m"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/global-local-information-soft-alignment-for","title":"Global–Local Information Soft-Alignment for Cross-Modal Remote-Sensing Image–Text Retrieval","date":"2024-05-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/efficient-remote-sensing-with-harmonized","title":"Efficient Remote Sensing with Harmonized Transfer Learning and Modality Alignment","date":"2024-04-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":1,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-prior-instruction-representation-framework","title":"A Prior Instruction Representation Framework for Remote Sensing Image-text Retrieval","date":"2023-10-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/direction-oriented-visual-semantic-embedding","title":"Direction-Oriented Visual-semantic Embedding Model for Remote Sensing Image-text Retrieval","date":"2023-10-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/parameter-efficient-transfer-learning-for-1","title":"Parameter-Efficient Transfer Learning for Remote Sensing Image-Text Retrieval","date":"2023-08-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rs5m-a-large-scale-vision-language-dataset","title":"RS5M and GeoRSCLIP: A Large Scale Vision-Language Dataset and A Large Vision-Language Model for Remote Sensing","date":"2023-06-20","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/remoteclip-a-vision-language-foundation-model","title":"RemoteCLIP: A Vision Language Foundation Model for Remote Sensing","date":"2023-06-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/reducing-semantic-confusion-scene-aware","title":"Reducing Semantic Confusion: Scene-aware Aggregation Network for Remote Sensing Cross-modal Retrieval","date":"2023-06-12","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/remote-sensing-cross-modal-text-image","title":"Remote Sensing Cross-Modal Text-Image Retrieval Based on Global and Local Information","date":"2022-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploring-a-fine-grained-multiscale-method","title":"Exploring a Fine-Grained Multiscale Method for Cross-Modal Remote Sensing Image Retrieval","date":"2022-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":16,"samples_ran":2,"samples_unverified":14,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}