{"url":"/dataset/rsitmd","name":"RSITMD","full_name":null,"description_markdown":"Click to add a brief description of the dataset (Markdown and LaTeX enabled).\r\n\r\nProvide:\r\n\r\n* a high-level explanation of the dataset characteristics\r\n* explain motivations and summary of its content\r\n* potential use cases of the dataset","description_withheld":null,"homepage":"","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/exploring-a-fine-grained-multiscale-method","title":"Exploring a Fine-Grained Multiscale Method for Cross-Modal Remote Sensing Image Retrieval","first_author":"Zhiqiang Yuan","url":null},"license":null,"modalities":[],"tasks":[{"name":"Cross-Modal Retrieval","url":"/task/cross-modal-retrieval","datasets_with_task":"/datasets/task/cross-modal-retrieval"}],"languages":[],"variants":["RSITMD"],"data_loaders":[],"num_papers_in_archive":26,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/cross-modal-retrieval-on-rsitmd","task":"Cross-Modal Retrieval","dataset_variant":"RSITMD","rows":10,"metrics":["Image-to-text R@1","Mean Recall","text-to-imageR@1"],"first_row_in_archive_order":{"model":"HarMA (w/ GeoRSCLIP)","paper":"/paper/efficient-remote-sensing-with-harmonized","metrics":{"Image-to-text R@1":"32.74%","Mean Recall":"52.27%","text-to-imageR@1":"25.62%"},"code_links":[{"title":"seekerhuang/harma","url":"https://github.com/seekerhuang/harma"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/global-local-information-soft-alignment-for","title":"Global–Local Information Soft-Alignment for Cross-Modal Remote-Sensing Image–Text Retrieval","date":"2024-05-14","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/efficient-remote-sensing-with-harmonized","title":"Efficient Remote Sensing with Harmonized Transfer Learning and Modality Alignment","date":"2024-04-28","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":1,"samples_unverified":5,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/a-prior-instruction-representation-framework","title":"A Prior Instruction Representation Framework for Remote Sensing Image-text Retrieval","date":"2023-10-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/direction-oriented-visual-semantic-embedding","title":"Direction-Oriented Visual-semantic Embedding Model for Remote Sensing Image-text Retrieval","date":"2023-10-12","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/parameter-efficient-transfer-learning-for-1","title":"Parameter-Efficient Transfer Learning for Remote Sensing Image-Text Retrieval","date":"2023-08-24","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/rs5m-a-large-scale-vision-language-dataset","title":"RS5M and GeoRSCLIP: A Large Scale Vision-Language Dataset and A Large Vision-Language Model for Remote Sensing","date":"2023-06-20","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":0,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/remoteclip-a-vision-language-foundation-model","title":"RemoteCLIP: A Vision Language Foundation Model for Remote Sensing","date":"2023-06-19","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/reducing-semantic-confusion-scene-aware","title":"Reducing Semantic Confusion: Scene-aware Aggregation Network for Remote Sensing Cross-modal Retrieval","date":"2023-06-12","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/remote-sensing-cross-modal-text-image","title":"Remote Sensing Cross-Modal Text-Image Retrieval Based on Global and Local Information","date":"2022-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":1,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploring-a-fine-grained-multiscale-method","title":"Exploring a Fine-Grained Multiscale Method for Cross-Modal Remote Sensing Image Retrieval","date":"2022-04-21","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":4,"samples_harvested":16,"samples_ran":2,"samples_unverified":14,"pointer_only_for_licence":1,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}