{"url":"/method/oscar","slug":"oscar","name":"OSCAR","full_name":"OSCAR","full_name_withheld":false,"description_markdown":"OSCAR is a new learning method that uses object tags detected in images as anchor points to ease the learning of image-text alignment. The model take a triple as input (word-tag-region) and pre-trained with two losses (masked token loss over words and tags, and a contrastive loss between tags and others). OSCAR represents an image-text pair into semantic space via dictionary lookup. Object tags are used as anchor points to align image regions with word embeddings of pre-trained language models. The model is then fine-tuned for understanding and generation tasks.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Oscar: Object-Semantics Aligned Pre-training for Vision-Language Tasks","paper":"/paper/oscar-object-semantics-aligned-pre-training","first_author":"Xiujun Li","n_authors":12,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/oscar-object-semantics-aligned-pre-training"},"source":{"url":"https://arxiv.org/abs/2004.06165v5","title":"Oscar: Object-Semantics Aligned Pre-training for Vision-Language Tasks","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision and Language Pre-Trained Models","url":"/methods/category/vision-and-language-pre-trained-models","pwc_aliases":[]}],"n_papers_tagged":36,"archive_num_papers":36,"papers_newest_first":[{"paper":"/paper/oscar-one-step-diffusion-codec-across","title":"OSCAR: One-Step Diffusion Codec for Image Compression Across Multiple Bit-rates","date":"2025-05-22","arxiv_id":"2505.16091","n_code_links":1,"syntology":null},{"paper":null,"title":"OSCAR: Online Soft Compression And Reranking","date":"2025-03-17","arxiv_id":"2504.07109","n_code_links":0,"syntology":null},{"paper":null,"title":"OSCAR: Object Status and Contextual Awareness for Recipes to Support Non-Visual Cooking","date":"2025-03-07","arxiv_id":"2503.05962","n_code_links":0,"syntology":null},{"paper":null,"title":"One-Shot Federated Learning with Classifier-Free Diffusion Models","date":"2025-02-12","arxiv_id":"2502.08488","n_code_links":0,"syntology":null},{"paper":null,"title":"Longitudinal Abuse and Sentiment Analysis of Hollywood Movie Dialogues using LLMs","date":"2025-01-20","arxiv_id":"2501.13948","n_code_links":0,"syntology":null},{"paper":null,"title":"Rephrasing natural text data with different languages and quality levels for Large Language Model pre-training","date":"2024-10-28","arxiv_id":"2410.20796","n_code_links":0,"syntology":null},{"paper":null,"title":"OSCAR: Operating System Control via State-Aware Reasoning and Re-Planning","date":"2024-10-24","arxiv_id":"2410.18963","n_code_links":0,"syntology":null},{"paper":null,"title":"Towards Fine-Grained Webpage Fingerprinting at Scale","date":"2024-09-06","arxiv_id":"2409.04341","n_code_links":0,"syntology":null},{"paper":null,"title":"IKUN for WMT24 General MT Task: LLMs Are here for Multilingual Machine Translation","date":"2024-08-21","arxiv_id":"2408.11512","n_code_links":0,"syntology":null},{"paper":"/paper/tropical-expressivity-of-neural-networks","title":"Tropical Expressivity of Neural Networks","date":"2024-05-30","arxiv_id":"2405.20174","n_code_links":1,"syntology":null},{"paper":"/paper/strong-screening-rules-for-group-based-slope","title":"Strong Screening Rules for Group-based SLOPE Models","date":"2024-05-24","arxiv_id":"2405.15357","n_code_links":1,"syntology":null},{"paper":null,"title":"Building a Large Japanese Web Corpus for Large Language Models","date":"2024-04-27","arxiv_id":"2404.17733","n_code_links":0,"syntology":null},{"paper":null,"title":"Automated Model Selection for Generalized Linear Models","date":"2024-04-25","arxiv_id":"2404.16560","n_code_links":0,"syntology":null},{"paper":null,"title":"Do Language Models Care About Text Quality? Evaluating Web-Crawled Corpora Across 11 Languages","date":"2024-03-13","arxiv_id":"2403.08693","n_code_links":0,"syntology":null},{"paper":"/paper/oscar-object-state-captioning-and-state","title":"OSCaR: Object State Captioning and State Change Representation","date":"2024-02-27","arxiv_id":"2402.17128","n_code_links":1,"syntology":null},{"paper":"/paper/glotscript-a-resource-and-tool-for-low","title":"GlotScript: A Resource and Tool for Low Resource Writing System Identification","date":"2023-09-23","arxiv_id":"2309.13320","n_code_links":1,"syntology":{"ran":0,"of":2,"unverified":2,"pointer_only":0}},{"paper":null,"title":"A Unified Framework for Pattern Recovery in Penalized and Thresholded Estimation and its Geometry","date":"2023-07-19","arxiv_id":"2307.10158","n_code_links":0,"syntology":null},{"paper":"/paper/one-stage-cascade-refinement-networks-for","title":"One-Stage Cascade Refinement Networks for Infrared Small Target Detection","date":"2022-12-16","arxiv_id":"2212.08472","n_code_links":2,"syntology":null},{"paper":null,"title":"RobBERT-2022: Updating a Dutch Language Model to Account for Evolving Language Use","date":"2022-11-15","arxiv_id":"2211.08192","n_code_links":0,"syntology":null},{"paper":"/paper/vl-checklist-evaluating-pre-trained-vision","title":"VL-CheckList: Evaluating Pre-trained Vision-Language Models with Objects, Attributes and Relations","date":"2022-07-01","arxiv_id":"2207.00221","n_code_links":1,"syntology":null},{"paper":null,"title":"Impact of Tokenization on Language Models: An Analysis for Turkish","date":"2022-04-19","arxiv_id":"2204.08832","n_code_links":0,"syntology":null},{"paper":null,"title":"WuDaoMM: A large-scale Multi-Modal Dataset for Pre-training models","date":"2022-03-22","arxiv_id":"2203.11480","n_code_links":0,"syntology":null},{"paper":null,"title":"Towards a Cleaner Document-Oriented Multilingual Crawled Corpus","date":"2022-01-17","arxiv_id":"2201.06642","n_code_links":0,"syntology":null},{"paper":null,"title":"Impact of Tokenization on Language Models: An Analysis for Turkish","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"PAGnol: An Extra-Large French Generative Model","date":"2021-10-16","arxiv_id":"2110.08554","n_code_links":0,"syntology":null},{"paper":"/paper/oscar-data-driven-operational-space-control","title":"OSCAR: Data-Driven Operational Space Control for Adaptive and Robust Robot Manipulation","date":"2021-10-02","arxiv_id":"2110.00704","n_code_links":1,"syntology":null},{"paper":null,"title":"Transliteration: A Simple Technique For Improving Multilingual Language Modeling","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"A Thorough Review on Recent Deep Learning Methodologies for Image Captioning","date":"2021-07-28","arxiv_id":"2107.13114","n_code_links":0,"syntology":null},{"paper":"/paper/perception-matters-detecting-perception","title":"Perception Matters: Detecting Perception Failures of VQA Models Using Metamorphic Testing","date":"2021-06-19","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/how-could-neural-networks-understand-programs","title":"How could Neural Networks understand Programs?","date":"2021-05-10","arxiv_id":"2105.04297","n_code_links":1,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/language-modeling","name":"Language Modeling","papers":5},{"task":"/task/language-modelling","name":"Language Modelling","papers":5},{"task":"/task/benchmarking","name":"Benchmarking","papers":3},{"task":"/task/image-captioning","name":"Image Captioning","papers":3},{"task":"/task/visual-question-answering","name":"Visual Question Answering (VQA)","papers":3},{"task":"/task/model","name":"model","papers":3},{"task":"/task/cg","name":"NER","papers":2},{"task":"/task/object","name":"Object","papers":2},{"task":"/task/question-answering","name":"Question Answering","papers":2},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":2},{"task":"/task/word-embeddings","name":"Word Embeddings","papers":2},{"task":"/task/feature-selection","name":"feature selection","papers":2},{"task":"/task/regression-1","name":"regression","papers":2},{"task":"/task/caption-generation","name":"Caption Generation","papers":1},{"task":"/task/change-detection","name":"Change Detection","papers":1},{"task":"/task/clustering","name":"Clustering","papers":1},{"task":"/task/cross-modal-retrieval","name":"Cross-Modal Retrieval","papers":1},{"task":"/task/dnn-testing","name":"DNN Testing","papers":1},{"task":"/task/dataset-generation","name":"Dataset Generation","papers":1},{"task":"/task/denoising","name":"Denoising","papers":1}],"tasks_shown":20,"n_tasks":55,"usage_by_year":[{"year":"2020","papers":5},{"year":"2021","papers":8},{"year":"2022","papers":6},{"year":"2023","papers":2},{"year":"2024","papers":10},{"year":"2025","papers":5}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/oscar"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}