{"url":"/method/galactica","slug":"galactica","name":"Galactica","full_name":"Galactica","full_name_withheld":false,"description_markdown":"Galactica is a language model which uses a Transformer architecture in a decoder-only setup with the following modifications:\r\n\r\n- It uses GeLU activations on all model sizes\r\n- It uses a 2048 length context window for all model sizes\r\n- It does not use biases in any of the dense kernels or layer norms\r\n- It uses learned positional embeddings for the model\r\n- A vocabulary of 50k tokens was constructed using BPE. The vocabulary was generated from a randomly selected 2% subset of the training data","description_state":"present","introduced_year":null,"introduced_by":{"title":"Galactica: A Large Language Model for Science","paper":"/paper/galactica-a-large-language-model-for-science-1","first_author":"Ross Taylor","n_authors":9,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/galactica-a-large-language-model-for-science-1"},"source":{"url":"https://arxiv.org/abs/2211.09085v1","title":"Galactica: A Large Language Model for Science","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Language Models","url":"/methods/category/language-models","pwc_aliases":[]}],"n_papers_tagged":10,"archive_num_papers":10,"papers_newest_first":[{"paper":"/paper/geogalactica-a-scientific-large-language","title":"GeoGalactica: A Scientific Large Language Model in Geoscience","date":"2023-12-31","arxiv_id":"2401.00434","n_code_links":1,"syntology":null},{"paper":"/paper/quality-quantity-synthetic-corpora-from","title":"TOP-Training: Target-Oriented Pretraining for Medical Extractive Question Answering","date":"2023-10-25","arxiv_id":"2310.16995","n_code_links":1,"syntology":null},{"paper":null,"title":"Unlocking Model Insights: A Dataset for Automated Model Card Generation","date":"2023-09-22","arxiv_id":"2309.12616","n_code_links":0,"syntology":null},{"paper":"/paper/2309-06256","title":"Mitigating the Alignment Tax of RLHF","date":"2023-09-12","arxiv_id":"2309.06256","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":null,"title":"Soft-prompt Tuning for Large Language Models to Evaluate Bias","date":"2023-06-07","arxiv_id":"2306.04735","n_code_links":0,"syntology":null},{"paper":"/paper/how-well-do-large-language-models-perform-in","title":"How well do Large Language Models perform in Arithmetic tasks?","date":"2023-03-16","arxiv_id":"2304.02015","n_code_links":1,"syntology":null},{"paper":null,"title":"Complex QA and language models hybrid architectures, Survey","date":"2023-02-17","arxiv_id":"2302.09051","n_code_links":0,"syntology":null},{"paper":null,"title":"ChatGPT versus Traditional Question Answering for Knowledge Graphs: Current Status and Future Directions Towards Knowledge Graph Chatbots","date":"2023-02-08","arxiv_id":"2302.06466","n_code_links":0,"syntology":null},{"paper":null,"title":"ChatGPT is not all you need. A State of the Art Review of large Generative AI models","date":"2023-01-11","arxiv_id":"2301.04655","n_code_links":0,"syntology":null},{"paper":"/paper/galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","arxiv_id":"2211.09085","n_code_links":1,"syntology":{"ran":0,"of":2,"unverified":2,"pointer_only":0}}],"papers_shown":10,"tasks":[{"task":"/task/question-answering","name":"Question Answering","papers":5},{"task":"/task/language-modeling","name":"Language Modeling","papers":3},{"task":"/task/language-modelling","name":"Language Modelling","papers":3},{"task":"/task/model","name":"model","papers":3},{"task":"/task/common-sense-reasoning","name":"Common Sense Reasoning","papers":2},{"task":"/task/domain-adaptation","name":"Domain Adaptation","papers":2},{"task":"/task/fairness","name":"Fairness","papers":2},{"task":"/task/large-language-model","name":"Large Language Model","papers":2},{"task":"/task/math","name":"Math","papers":2},{"task":"/task/all","name":"All","papers":1},{"task":"/task/anachronisms","name":"Anachronisms","papers":1},{"task":"/task/bias-detection","name":"Bias Detection","papers":1},{"task":"/task/chatbot","name":"Chatbot","papers":1},{"task":"/task/citation-prediction","name":"Citation Prediction","papers":1},{"task":"/task/classification-1","name":"Classification","papers":1},{"task":"/task/continual-learning","name":"Continual Learning","papers":1},{"task":"/task/document-classification","name":"Document Classification","papers":1},{"task":"/task/extractive-question-answering","name":"Extractive Question-Answering","papers":1},{"task":"/task/general-knowledge","name":"General Knowledge","papers":1},{"task":"/task/iupac-name-prediction","name":"IUPAC Name Prediction","papers":1}],"tasks_shown":20,"n_tasks":43,"usage_by_year":[{"year":"2022","papers":1},{"year":"2023","papers":9}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/galactica"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}