{"url":"/task/image-to-text","name":"Image to text","slug":"image-to-text","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":246,"papers_with_code":103,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/cii-bench","name":"CII-Bench","full_name":"Chinese Image Implication understanding Benchmark","num_papers_in_archive":2},{"url":"/dataset/psocr","name":"PsOCR","full_name":"Pashto OCR Dataset","num_papers_in_archive":1},{"url":"/dataset/urdu-text-scene-images","name":"Urdu Text Scene Images","full_name":"","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":103,"tagged_in_all":246,"items":[{"url":"/paper/blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","repositories_listed":17,"syntology":{"n":8,"n_ran":4,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/pix2struct-screenshot-parsing-as-pretraining","title":"Pix2Struct: Screenshot Parsing as Pretraining for Visual Language Understanding","date":"2022-10-07","arxiv_id":"2210.03347","repositories_listed":4,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/distilled-dual-encoder-model-for-vision","title":"Distilled Dual-Encoder Model for Vision-Language Understanding","date":"2021-12-16","arxiv_id":"2112.08723","repositories_listed":4,"syntology":null},{"url":"/paper/effective-use-of-word-order-for-text-1","title":"Effective Use of Word Order for Text Categorization with Convolutional Neural Networks","date":"2014-12-01","arxiv_id":"1412.1058","repositories_listed":4,"syntology":null},{"url":"/paper/cephalo-multi-modal-vision-language-models","title":"Cephalo: Multi-Modal Vision-Language Models for Bio-Inspired Materials Analysis and Design","date":"2024-05-29","arxiv_id":"2405.19076","repositories_listed":3,"syntology":null},{"url":"/paper/evaluating-text-to-visual-generation-with","title":"Evaluating Text-to-Visual Generation with Image-to-Text Generation","date":"2024-04-01","arxiv_id":"2404.01291","repositories_listed":3,"syntology":null},{"url":"/paper/one-transformer-fits-all-distributions-in","title":"One Transformer Fits All Distributions in Multi-Modal Diffusion at Scale","date":"2023-03-12","arxiv_id":"2303.06555","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/versatile-diffusion-text-images-and","title":"Versatile Diffusion: Text, Images and Variations All in One Diffusion Model","date":"2022-11-15","arxiv_id":"2211.08332","repositories_listed":3,"syntology":null},{"url":"/paper/improving-factual-completeness-and","title":"Improving Factual Completeness and Consistency of Image-to-Text Radiology Report Generation","date":"2020-10-20","arxiv_id":"2010.10042","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/lmm4lmm-benchmarking-and-evaluating-large","title":"LMM4LMM: Benchmarking and Evaluating Large-multimodal Image Generation with LMMs","date":"2025-04-11","arxiv_id":"2504.08358","repositories_listed":2,"syntology":null},{"url":"/paper/magma-a-foundation-model-for-multimodal-ai","title":"Magma: A Foundation Model for Multimodal AI Agents","date":"2025-02-18","arxiv_id":"2502.13130","repositories_listed":2,"syntology":{"n":16,"n_ran":2,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/survey-on-abstractive-text-summarization","title":"Survey on Abstractive Text Summarization: Dataset, Models, and Metrics","date":"2024-12-22","arxiv_id":"2412.17165","repositories_listed":2,"syntology":null},{"url":"/paper/semantic-editing-increment-benefits-zero-shot","title":"Semantic Editing Increment Benefits Zero-Shot Composed Image Retrieval","date":"2024-10-28","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/ldre-llm-based-divergent-reasoning-and","title":"LDRE: LLM-based Divergent Reasoning and Ensemble for Zero-Shot Composed Image Retrieval","date":"2024-07-11","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/comat-aligning-text-to-image-diffusion-model","title":"CoMat: Aligning Text-to-Image Diffusion Model with Image-to-Text Concept Matching","date":"2024-04-04","arxiv_id":"2404.03653","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_unverified":4,"n_pointer_only":9}},{"url":"/paper/large-multilingual-models-pivot-zero-shot","title":"Large Multilingual Models Pivot Zero-Shot Multimodal Learning across Languages","date":"2023-08-23","arxiv_id":"2308.12038","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_unverified":1,"n_pointer_only":7}},{"url":"/paper/multimodal-dataset-distillation-for-image","title":"Vision-Language Dataset Distillation","date":"2023-08-15","arxiv_id":"2308.07545","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/generative-pretraining-in-multimodality","title":"Emu: Generative Pretraining in Multimodality","date":"2023-07-11","arxiv_id":"2307.05222","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/zeronlg-aligning-and-autoencoding-domains-for","title":"ZeroNLG: Aligning and Autoencoding Domains for Zero-Shot Multimodal and Multilingual Natural Language Generation","date":"2023-03-11","arxiv_id":"2303.06458","repositories_listed":2,"syntology":null},{"url":"/paper/safe-latent-diffusion-mitigating","title":"Safe Latent Diffusion: Mitigating Inappropriate Degeneration in Diffusion Models","date":"2022-11-09","arxiv_id":"2211.05105","repositories_listed":2,"syntology":null},{"url":"/paper/linearly-mapping-from-image-to-text-space","title":"Linearly Mapping from Image to Text Space","date":"2022-09-30","arxiv_id":"2209.15162","repositories_listed":2,"syntology":{"n":14,"n_ran":5,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/ernie-vilg-unified-generative-pre-training","title":"ERNIE-ViLG: Unified Generative Pre-training for Bidirectional Vision-Language Generation","date":"2021-12-31","arxiv_id":"2112.15283","repositories_listed":2,"syntology":null},{"url":"/paper/translation-equivariant-image-quantizer-for","title":"Exploration into Translation-Equivariant Image Quantization","date":"2021-12-01","arxiv_id":"2112.00384","repositories_listed":2,"syntology":{"n":14,"n_ran":13,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/mirrorgan-learning-text-to-image-generation","title":"MirrorGAN: Learning Text-to-image Generation by Redescription","date":"2019-03-14","arxiv_id":"1903.05854","repositories_listed":2,"syntology":null},{"url":"/paper/text-to-image-to-text-translation-using-cycle","title":"Text-to-Image-to-Text Translation using Cycle Consistent Adversarial Networks","date":"2018-08-14","arxiv_id":"1808.04538","repositories_listed":2,"syntology":null},{"url":"/paper/efficient-medical-vision-language-alignment","title":"Efficient Medical Vision-Language Alignment Through Adapting Masked Vision Models","date":"2025-06-10","arxiv_id":"2506.08990","repositories_listed":1,"syntology":null},{"url":"/paper/unimoco-unified-modality-completion-for","title":"UniMoCo: Unified Modality Completion for Robust Multi-Modal Embeddings","date":"2025-05-17","arxiv_id":"2505.11815","repositories_listed":1,"syntology":null},{"url":"/paper/lrsclip-a-vision-language-foundation-model","title":"LRSCLIP: A Vision-Language Foundation Model for Aligning Remote Sensing Image with Longer Text","date":"2025-03-25","arxiv_id":"2503.19311","repositories_listed":1,"syntology":null},{"url":"/paper/real-world-validation-of-a-multimodal-llm","title":"Real-world validation of a multimodal LLM-powered pipeline for High-Accuracy Clinical Trial Patient Matching leveraging EHR data","date":"2025-03-19","arxiv_id":"2503.15374","repositories_listed":1,"syntology":null},{"url":"/paper/flowtok-flowing-seamlessly-across-text-and","title":"FlowTok: Flowing Seamlessly Across Text and Image Tokens","date":"2025-03-13","arxiv_id":"2503.10772","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_unverified":3,"n_pointer_only":0}}],"syntology_records":12,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}