{"url":"/method/virtex","slug":"virtex","name":"VirTex","full_name":"VirTex","full_name_withheld":false,"description_markdown":"**VirText**, or **Visual representations from Textual annotations** is a pretraining approach using semantically dense captions to learn visual representations. First a ConvNet and [Transformer](https://paperswithcode.com/method/transformer) are jointly trained from scratch to generate natural language captions for images. Then, the learned features are transferred to downstream visual recognition tasks.","description_state":"present","introduced_year":null,"introduced_by":{"title":"VirTex: Learning Visual Representations from Textual Annotations","paper":"/paper/virtex-learning-visual-representations-from","first_author":"Karan Desai","n_authors":2,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/virtex-learning-visual-representations-from"},"source":{"url":"https://arxiv.org/abs/2006.06666v3","title":"VirTex: Learning Visual Representations from Textual Annotations","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Image Representations","url":"/methods/category/image-representations","pwc_aliases":[]}],"n_papers_tagged":6,"archive_num_papers":6,"papers_newest_first":[{"paper":"/paper/hlstransform-energy-efficient-llama-2","title":"HLSTransform: Energy-Efficient Llama 2 Inference on FPGAs Via High Level Synthesis","date":"2024-04-29","arxiv_id":"2405.00738","n_code_links":1,"syntology":null},{"paper":null,"title":"Chaos-Based Bitwise Dynamical Pseudorandom Number Generator on FPGA","date":"2024-01-26","arxiv_id":"2401.15035","n_code_links":0,"syntology":null},{"paper":null,"title":"Real-time FPGA Design for OMP Targeting 8K Image Reconstruction","date":"2021-10-10","arxiv_id":"2110.04714","n_code_links":0,"syntology":null},{"paper":null,"title":"Reconfigurable co-processor architecture with limited numerical precision to accelerate deep convolutional neural networks","date":"2021-08-21","arxiv_id":"2109.03040","n_code_links":0,"syntology":null},{"paper":null,"title":"FPGA Implementation of Simplified Spiking Neural Network","date":"2020-10-02","arxiv_id":"2010.01200","n_code_links":0,"syntology":null},{"paper":"/paper/virtex-learning-visual-representations-from","title":"VirTex: Learning Visual Representations from Textual Annotations","date":"2020-06-11","arxiv_id":"2006.06666","n_code_links":3,"syntology":{"ran":0,"of":1,"unverified":1,"pointer_only":0}}],"papers_shown":6,"tasks":[{"task":null,"name":"8k","papers":1},{"task":null,"name":"CPU","papers":1},{"task":"/task/edge-computing","name":"Edge-computing","papers":1},{"task":null,"name":"GPU","papers":1},{"task":"/task/classification","name":"General Classification","papers":1},{"task":"/task/high-level-synthesis","name":"High-Level Synthesis","papers":1},{"task":"/task/image-captioning","name":"Image Captioning","papers":1},{"task":"/task/image-classification","name":"Image Classification","papers":1},{"task":"/task/image-reconstruction","name":"Image Reconstruction","papers":1},{"task":"/task/instance-segmentation","name":"Instance Segmentation","papers":1},{"task":"/task/object-detection","name":"Object Detection","papers":1},{"task":"/task/quantization","name":"Quantization","papers":1},{"task":"/task/semantic-segmentation","name":"Semantic Segmentation","papers":1},{"task":"/task/compressed-sensing","name":"compressed sensing","papers":1},{"task":"/task/image-classification","name":"image-classification","papers":1},{"task":"/task/object-detection-1","name":"object-detection","papers":1}],"tasks_shown":16,"n_tasks":16,"usage_by_year":[{"year":"2020","papers":2},{"year":"2021","papers":2},{"year":"2024","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/virtex"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}