{"url":"/method/florence","slug":"florence","name":"Florence","full_name":"Florence","full_name_withheld":false,"description_markdown":"Florence is a computer vision foundation model aiming to learn universal visual-language representations that be adapted to various computer vision tasks, visual question answering, image captioning, video retrieval, among other tasks. Florence's workflow consists of data curation, unified learning, Transformer architectures and adaption. Florence is pre-trained in an image-label-description space, utilizing a unified image-text contrastive learning. It involves a two-tower architecture: 12-layer Transformer for the language encoder, and a Vision Transformer for the image encoder. Two linear projection layers are added on top of the image encoder and language encoder to match the dimensions of image and language features. Compared to previous methods for cross-modal shared representations, Florence expands beyond simple classification and retrieval capabilities to advanced representations that support object level, multiple modality, and videos respectively.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Florence: A New Foundation Model for Computer Vision","paper":"/paper/florence-a-new-foundation-model-for-computer","first_author":"Lu Yuan","n_authors":23,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/florence-a-new-foundation-model-for-computer"},"source":{"url":"https://arxiv.org/abs/2111.11432v1","title":"Florence: A New Foundation Model for Computer Vision","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision and Language Pre-Trained Models","url":"/methods/category/vision-and-language-pre-trained-models","pwc_aliases":[]}],"n_papers_tagged":12,"archive_num_papers":12,"papers_newest_first":[{"paper":null,"title":"FRED: The Florence RGB-Event Drone Dataset","date":"2025-06-05","arxiv_id":"2506.05163","n_code_links":0,"syntology":null},{"paper":"/paper/psocr-benchmarking-large-multimodal-models","title":"PsOCR: Benchmarking Large Multimodal Models for Optical Character Recognition in Low-resource Pashto Language","date":"2025-05-15","arxiv_id":"2505.10055","n_code_links":1,"syntology":null},{"paper":null,"title":"Quantifying walkable accessibility to urban services: An application to Florence, Italy","date":"2025-04-17","arxiv_id":"2504.12934","n_code_links":0,"syntology":null},{"paper":null,"title":"Fine-Tuning Florence2 for Enhanced Object Detection in Un-constructed Environments: Vision-Language Model Approach","date":"2025-03-06","arxiv_id":"2503.04918","n_code_links":0,"syntology":null},{"paper":"/paper/texlidar-automated-text-understanding-for","title":"TexLiDAR: Automated Text Understanding for Panoramic LiDAR Data","date":"2025-02-05","arxiv_id":"2502.04385","n_code_links":1,"syntology":null},{"paper":null,"title":"Learning Nonverbal Cues in Multiparty Social Interactions for Robotic Facilitators","date":"2025-01-18","arxiv_id":"2501.10857","n_code_links":0,"syntology":null},{"paper":null,"title":"Smart City Digital Twin Framework for Real-Time Multi-Data Integration and Wide Public Distribution","date":"2023-09-23","arxiv_id":"2309.13394","n_code_links":0,"syntology":null},{"paper":"/paper/asm-adaptive-skinning-model-for-high-quality","title":"ASM: Adaptive Skinning Model for High-Quality 3D Face Modeling","date":"2023-04-19","arxiv_id":"2304.09423","n_code_links":0,"syntology":null},{"paper":null,"title":"DIME-FM: DIstilling Multimodal and Efficient Foundation Models","date":"2023-03-31","arxiv_id":"2303.18232","n_code_links":0,"syntology":null},{"paper":null,"title":"DIME-FM : DIstilling Multimodal and Efficient Foundation Models","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/the-florence-4d-facial-expression-dataset","title":"The Florence 4D Facial Expression Dataset","date":"2022-10-30","arxiv_id":"2210.16807","n_code_links":0,"syntology":null},{"paper":"/paper/florence-a-new-foundation-model-for-computer","title":"Florence: A New Foundation Model for Computer Vision","date":"2021-11-22","arxiv_id":"2111.11432","n_code_links":2,"syntology":null}],"papers_shown":12,"tasks":[{"task":"/task/image-classification","name":"Image Classification","papers":3},{"task":"/task/object-detection","name":"Object Detection","papers":3},{"task":"/task/object-detection-1","name":"object-detection","papers":3},{"task":"/task/benchmarking","name":"Benchmarking","papers":2},{"task":"/task/image-classification","name":"image-classification","papers":2},{"task":"/task/3d-face-reconstruction","name":"3D Face Reconstruction","papers":1},{"task":"/task/action-classification","name":"Action Classification","papers":1},{"task":"/task/action-recognition-in-videos","name":"Action Recognition","papers":1},{"task":"/task/action-recognition-in-videos-2","name":"Action Recognition In Videos","papers":1},{"task":"/task/cross-modal-retrieval","name":"Cross-Modal Retrieval","papers":1},{"task":"/task/data-integration","name":"Data Integration","papers":1},{"task":"/task/face-alignment","name":"Face Alignment","papers":1},{"task":"/task/face-model","name":"Face Model","papers":1},{"task":"/task/face-reconstruction","name":"Face Reconstruction","papers":1},{"task":null,"name":"GPU","papers":1},{"task":"/task/image-captioning","name":"Image Captioning","papers":1},{"task":"/task/language-modeling","name":"Language Modeling","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/object","name":"Object","papers":1},{"task":"/task/optical-character-recognition","name":"Optical Character Recognition","papers":1}],"tasks_shown":20,"n_tasks":36,"usage_by_year":[{"year":"2021","papers":1},{"year":"2022","papers":1},{"year":"2023","papers":4},{"year":"2025","papers":6}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/florence"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}