{"url":"/dataset/big-bench","name":"BIG-bench","full_name":"Beyond the Imitation Game Benchmark","description_markdown":"The **Beyond the Imitation Game Benchmark** (BIG-bench) is a collaborative benchmark intended to probe large language models and extrapolate their future capabilities. Big-bench include more than 200 tasks.\r\n\r\nImage source: [https://arxiv.org/pdf/2206.04615.pdf](https://arxiv.org/pdf/2206.04615.pdf)","description_withheld":null,"homepage":"https://github.com/google/BIG-bench","introduced_date":"2022-06-09","introduced_date_note":null,"introduced_by":{"paper":"/paper/beyond-the-imitation-game-quantifying-and","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","first_author":"Aarohi Srivastava","url":null},"license":{"name":"Apache License 2.0","url":"https://github.com/google/BIG-bench/blob/main/LICENSE"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"task","url":null,"datasets_with_task":"/datasets/task/task"},{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"},{"name":"Common Sense Reasoning","url":"/task/common-sense-reasoning","datasets_with_task":"/datasets/task/common-sense-reasoning"},{"name":"Multiple Choice Question Answering (MCQA)","url":"/task/multiple-choice-qa","datasets_with_task":"/datasets/task/multiple-choice-qa"},{"name":"Logical Reasoning","url":"/task/logical-reasoning","datasets_with_task":"/datasets/task/logical-reasoning"},{"name":"Word Sense Disambiguation","url":"/task/word-sense-disambiguation","datasets_with_task":"/datasets/task/word-sense-disambiguation"},{"name":"Sarcasm Detection","url":"/task/sarcasm-detection","datasets_with_task":"/datasets/task/sarcasm-detection"},{"name":"General Knowledge","url":"/task/general-knowledge","datasets_with_task":"/datasets/task/general-knowledge"},{"name":"Multi-task Language Understanding","url":"/task/multi-task-language-understanding","datasets_with_task":"/datasets/task/multi-task-language-understanding"},{"name":"Intent Recognition","url":"/task/intent-recognition","datasets_with_task":"/datasets/task/intent-recognition"},{"name":"BIG-bench Machine Learning","url":"/task/machine-learning","datasets_with_task":"/datasets/task/machine-learning"},{"name":"Riddle Sense","url":"/task/riddle-sense","datasets_with_task":"/datasets/task/riddle-sense"},{"name":"Natural Questions","url":"/task/natural-questions","datasets_with_task":"/datasets/task/natural-questions"},{"name":"Analogical Similarity","url":"/task/analogical-similarity","datasets_with_task":"/datasets/task/analogical-similarity"},{"name":"Identify Odd Metapor","url":"/task/identify-odd-metapor","datasets_with_task":"/datasets/task/identify-odd-metapor"},{"name":"Odd One Out","url":"/task/odd-one-out","datasets_with_task":"/datasets/task/odd-one-out"},{"name":"Crash Blossom","url":"/task/crash-blossom","datasets_with_task":"/datasets/task/crash-blossom"},{"name":"Auto Debugging","url":"/task/auto-debugging","datasets_with_task":"/datasets/task/auto-debugging"},{"name":"Crass AI","url":"/task/crass-ai","datasets_with_task":"/datasets/task/crass-ai"},{"name":"Discourse Marker Prediction","url":"/task/discourse-marker-prediction","datasets_with_task":"/datasets/task/discourse-marker-prediction"},{"name":"Empirical Judgments","url":"/task/empirical-judgments","datasets_with_task":"/datasets/task/empirical-judgments"},{"name":"Irony Identification","url":"/task/irony-identification","datasets_with_task":"/datasets/task/irony-identification"},{"name":"Timedial","url":"/task/timedial","datasets_with_task":"/datasets/task/timedial"},{"name":"Understanding Fables","url":"/task/understanding-fables","datasets_with_task":"/datasets/task/understanding-fables"},{"name":"Dark Humor Detection","url":"/task/dark-humor-detection","datasets_with_task":"/datasets/task/dark-humor-detection"},{"name":"Business Ethics","url":"/task/business-ethics","datasets_with_task":"/datasets/task/business-ethics"},{"name":"Moral Disputes","url":"/task/moral-disputes","datasets_with_task":"/datasets/task/moral-disputes"},{"name":"Moral Permissibility","url":"/task/moral-permissibility","datasets_with_task":"/datasets/task/moral-permissibility"},{"name":"Moral Scenarios","url":"/task/moral-scenarios","datasets_with_task":"/datasets/task/moral-scenarios"},{"name":"FEVER (2-way)","url":"/task/fever-2-way","datasets_with_task":"/datasets/task/fever-2-way"},{"name":"FEVER (3-way)","url":"/task/fever-3-way","datasets_with_task":"/datasets/task/fever-3-way"},{"name":"Misconceptions","url":"/task/misconceptions","datasets_with_task":"/datasets/task/misconceptions"},{"name":"Sentence Ambiguity","url":"/task/sentence-ambiguity","datasets_with_task":"/datasets/task/sentence-ambiguity"},{"name":"Global Facts","url":"/task/global-facts","datasets_with_task":"/datasets/task/global-facts"},{"name":"Miscellaneous","url":"/task/miscellaneous","datasets_with_task":"/datasets/task/miscellaneous"},{"name":"Similarities Abstraction","url":"/task/similarities-abstraction","datasets_with_task":"/datasets/task/similarities-abstraction"},{"name":"TriviaQA","url":"/task/triviaqa","datasets_with_task":"/datasets/task/triviaqa"},{"name":"High School European History","url":"/task/high-school-european-history","datasets_with_task":"/datasets/task/high-school-european-history"},{"name":"High School US History","url":"/task/high-school-us-history","datasets_with_task":"/datasets/task/high-school-us-history"},{"name":"High School World History","url":"/task/high-school-world-history","datasets_with_task":"/datasets/task/high-school-world-history"},{"name":"International Law","url":"/task/international-law","datasets_with_task":"/datasets/task/international-law"},{"name":"Jurisprudence","url":"/task/jurisprudence","datasets_with_task":"/datasets/task/jurisprudence"},{"name":"Logical Fallacies","url":"/task/logical-fallacies","datasets_with_task":"/datasets/task/logical-fallacies"},{"name":"Management","url":"/task/management","datasets_with_task":"/datasets/task/management"},{"name":"Marketing","url":"/task/marketing","datasets_with_task":"/datasets/task/marketing"},{"name":"Philosophy","url":"/task/philosophy","datasets_with_task":"/datasets/task/philosophy"},{"name":"Prehistory","url":"/task/prehistory","datasets_with_task":"/datasets/task/prehistory"},{"name":"Professional Law","url":"/task/professional-law","datasets_with_task":"/datasets/task/professional-law"},{"name":"World Religions","url":"/task/world-religions","datasets_with_task":"/datasets/task/world-religions"},{"name":"Analytic Entailment","url":"/task/analytic-entailment","datasets_with_task":"/datasets/task/analytic-entailment"},{"name":"Entailed Polarity","url":"/task/entailed-polarity","datasets_with_task":"/datasets/task/entailed-polarity"},{"name":"Epistemic Reasoning","url":"/task/epistemic-reasoning","datasets_with_task":"/datasets/task/epistemic-reasoning"},{"name":"Evaluating Information Essentiality","url":"/task/evaluating-information-essentiality","datasets_with_task":"/datasets/task/evaluating-information-essentiality"},{"name":"Logical Args","url":"/task/logical-args","datasets_with_task":"/datasets/task/logical-args"},{"name":"Metaphor Boolean","url":"/task/metaphor-boolean","datasets_with_task":"/datasets/task/metaphor-boolean"},{"name":"Physical Intuition","url":"/task/physical-intuition","datasets_with_task":"/datasets/task/physical-intuition"},{"name":"Presuppositions As NLI","url":"/task/presuppositions-as-nli","datasets_with_task":"/datasets/task/presuppositions-as-nli"},{"name":"Abstract Algebra","url":"/task/abstract-algebra","datasets_with_task":"/datasets/task/abstract-algebra"},{"name":"College Mathematics","url":"/task/college-mathematics","datasets_with_task":"/datasets/task/college-mathematics"},{"name":"Elementary Mathematics","url":"/task/elementary-mathematics","datasets_with_task":"/datasets/task/elementary-mathematics"},{"name":"Formal Logic","url":"/task/formal-logic","datasets_with_task":"/datasets/task/formal-logic"},{"name":"High School Mathematics","url":"/task/high-school-mathematics","datasets_with_task":"/datasets/task/high-school-mathematics"},{"name":"Mathematical Induction","url":"/task/mathematical-induction","datasets_with_task":"/datasets/task/mathematical-induction"},{"name":"Professional Accounting","url":"/task/professional-accounting","datasets_with_task":"/datasets/task/professional-accounting"},{"name":"Anatomy","url":"/task/anatomy","datasets_with_task":"/datasets/task/anatomy"},{"name":"Clinical Knowledge","url":"/task/clinical-knowledge","datasets_with_task":"/datasets/task/clinical-knowledge"},{"name":"College Medicine","url":"/task/college-medicine","datasets_with_task":"/datasets/task/college-medicine"},{"name":"Human Aging","url":"/task/human-aging","datasets_with_task":"/datasets/task/human-aging"},{"name":"Human Organs Senses Multiple Choice","url":"/task/human-organs-senses-multiple-choice","datasets_with_task":"/datasets/task/human-organs-senses-multiple-choice"},{"name":"Medical Genetics","url":"/task/medical-genetics","datasets_with_task":"/datasets/task/medical-genetics"},{"name":"Nutrition","url":"/task/nutrition","datasets_with_task":"/datasets/task/nutrition"},{"name":"Professional Medicine","url":"/task/professional-medicine","datasets_with_task":"/datasets/task/professional-medicine"},{"name":"Virology","url":"/task/virology","datasets_with_task":"/datasets/task/virology"},{"name":"English Proverbs","url":"/task/english-proverbs","datasets_with_task":"/datasets/task/english-proverbs"},{"name":"Fantasy Reasoning","url":"/task/fantasy-reasoning","datasets_with_task":"/datasets/task/fantasy-reasoning"},{"name":"Figure Of Speech Detection","url":"/task/figure-of-speech-detection","datasets_with_task":"/datasets/task/figure-of-speech-detection"},{"name":"GRE Reading Comprehension","url":"/task/gre-reading-comprehension","datasets_with_task":"/datasets/task/gre-reading-comprehension"},{"name":"Implicatures","url":"/task/implicatures","datasets_with_task":"/datasets/task/implicatures"},{"name":"Implicit Relations","url":"/task/implicit-relations","datasets_with_task":"/datasets/task/implicit-relations"},{"name":"LAMBADA","url":"/task/lambada","datasets_with_task":"/datasets/task/lambada"},{"name":"Movie Dialog Same Or Different","url":"/task/movie-dialog-same-or-different","datasets_with_task":"/datasets/task/movie-dialog-same-or-different"},{"name":"Nonsense Words Grammar","url":"/task/nonsense-words-grammar","datasets_with_task":"/datasets/task/nonsense-words-grammar"},{"name":"Phrase Relatedness","url":"/task/phrase-relatedness","datasets_with_task":"/datasets/task/phrase-relatedness"},{"name":"Question Selection","url":"/task/question-selection","datasets_with_task":"/datasets/task/question-selection"},{"name":"RACE-h","url":"/task/race-h","datasets_with_task":"/datasets/task/race-h"},{"name":"RACE-m","url":"/task/race-m","datasets_with_task":"/datasets/task/race-m"},{"name":"Astronomy","url":"/task/astronomy","datasets_with_task":"/datasets/task/astronomy"},{"name":"College Biology","url":"/task/college-biology","datasets_with_task":"/datasets/task/college-biology"},{"name":"College Chemistry","url":"/task/college-chemistry","datasets_with_task":"/datasets/task/college-chemistry"},{"name":"College Computer Science","url":"/task/college-computer-science","datasets_with_task":"/datasets/task/college-computer-science"},{"name":"College Physics","url":"/task/college-physics","datasets_with_task":"/datasets/task/college-physics"},{"name":"Computer Security","url":"/task/computer-security","datasets_with_task":"/datasets/task/computer-security"},{"name":"Conceptual Physics","url":"/task/conceptual-physics","datasets_with_task":"/datasets/task/conceptual-physics"},{"name":"Electrical Engineering","url":"/task/electrical-engineering","datasets_with_task":"/datasets/task/electrical-engineering"},{"name":"High School Biology","url":"/task/high-school-biology","datasets_with_task":"/datasets/task/high-school-biology"},{"name":"High School Chemistry","url":"/task/high-school-chemistry","datasets_with_task":"/datasets/task/high-school-chemistry"},{"name":"High School Computer Science","url":"/task/high-school-computer-science","datasets_with_task":"/datasets/task/high-school-computer-science"},{"name":"High School Physics","url":"/task/high-school-physics","datasets_with_task":"/datasets/task/high-school-physics"},{"name":"High School Statistics","url":"/task/high-school-statistics","datasets_with_task":"/datasets/task/high-school-statistics"},{"name":"Physics MC","url":"/task/physics-mc","datasets_with_task":"/datasets/task/physics-mc"},{"name":"Econometrics","url":"/task/econometrics","datasets_with_task":"/datasets/task/econometrics"},{"name":"High School Geography","url":"/task/high-school-geography","datasets_with_task":"/datasets/task/high-school-geography"},{"name":"High School Government and Politics","url":"/task/high-school-government-and-politics","datasets_with_task":"/datasets/task/high-school-government-and-politics"},{"name":"High School Macroeconomics","url":"/task/high-school-macroeconomics","datasets_with_task":"/datasets/task/high-school-macroeconomics"},{"name":"High School Microeconomics","url":"/task/high-school-microeconomics","datasets_with_task":"/datasets/task/high-school-microeconomics"},{"name":"High School Psychology","url":"/task/high-school-psychology","datasets_with_task":"/datasets/task/high-school-psychology"},{"name":"Human Sexuality","url":"/task/human-sexuality","datasets_with_task":"/datasets/task/human-sexuality"},{"name":"Professional Psychology","url":"/task/professional-psychology","datasets_with_task":"/datasets/task/professional-psychology"},{"name":"Public Relations","url":"/task/public-relations","datasets_with_task":"/datasets/task/public-relations"},{"name":"Security Studies","url":"/task/security-studies","datasets_with_task":"/datasets/task/security-studies"},{"name":"Sociology","url":"/task/sociology","datasets_with_task":"/datasets/task/sociology"},{"name":"US Foreign Policy","url":"/task/us-foreign-policy","datasets_with_task":"/datasets/task/us-foreign-policy"},{"name":"Memorization","url":"/task/memorization","datasets_with_task":"/datasets/task/memorization"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["BBH-nlp","BBH-alg","Big-bench Hard","BIG-bench (Logical Sequence)","BIG-bench (Logical Fallacy Detection)","BIG-bench (Known Unknowns)","BIG-bench (Hindu Knowledge)","BIG-bench (Novel Concepts)","BIG-bench (StrategyQA)","BIG-bench (Winowhy)","BIG-bench (Logic Grid Puzzle)","BIG-bench (Anachronisms)","BIG-bench (Temporal Sequences)","BIG-bench (Sports Understanding)","BIG-bench (SNARKS)","BIG-bench (Ruin Names)","BIG-bench (Reasoning About Colored Objects)","BIG-bench (Penguins In A Table)","BIG-bench (Navigate)","BIG-bench (Movie Recommendation)","BIG-bench (Hyperbaton)","BIG-bench (Formal Fallacies Syllogisms Negation)","BIG-bench (Disambiguation QA)","BIG-bench (Date Understanding)","BIG-bench (Causal Judgment)","BIG-bench-lite","Big-bench Lite","BIG-bench"],"data_loaders":[],"num_papers_in_archive":349,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/multi-task-language-understanding-on-bbh-nlp","task":"Multi-task Language Understanding","dataset_variant":"BBH-nlp","rows":15,"metrics":["Average (%)"],"first_row_in_archive_order":{"model":"Qwen2.5-72B","paper":null,"metrics":{"Average (%)":"86.3"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/common-sense-reasoning-on-big-bench","task":"Common Sense Reasoning","dataset_variant":"BIG-bench (Disambiguation QA)","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=3, Direct)","paper":"/paper/palm-2-technical-report-1","metrics":{"Accuracy":"78.8"},"code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/common-sense-reasoning-on-big-bench-causal","task":"Common Sense Reasoning","dataset_variant":"BIG-bench (Causal Judgment)","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=3, Direct)","paper":"/paper/palm-2-technical-report-1","metrics":{"Accuracy":"62.0"},"code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/common-sense-reasoning-on-big-bench-date","task":"Common Sense Reasoning","dataset_variant":"BIG-bench (Date Understanding)","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=3, CoT)","paper":"/paper/palm-2-technical-report-1","metrics":{"Accuracy":"91.2"},"code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/logical-reasoning-on-big-bench-formal","task":"Logical Reasoning","dataset_variant":"BIG-bench (Formal Fallacies Syllogisms Negation)","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=3, Direct)","paper":"/paper/palm-2-technical-report-1","metrics":{"Accuracy":"64.8"},"code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/logical-reasoning-on-big-bench-penguins-in-a","task":"Logical Reasoning","dataset_variant":"BIG-bench (Penguins In A Table)","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=3, CoT)","paper":"/paper/palm-2-technical-report-1","metrics":{"Accuracy":"84.9"},"code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/logical-reasoning-on-big-bench-reasoning","task":"Logical Reasoning","dataset_variant":"BIG-bench (Reasoning About Colored Objects)","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=3, CoT)","paper":"/paper/palm-2-technical-report-1","metrics":{"Accuracy":"91.2"},"code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/logical-reasoning-on-big-bench-temporal","task":"Logical Reasoning","dataset_variant":"BIG-bench (Temporal Sequences)","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=3, CoT)","paper":"/paper/palm-2-technical-report-1","metrics":{"Accuracy":"100"},"code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-27","task":"Multiple Choice Question Answering (MCQA)","dataset_variant":"BIG-bench (Hyperbaton)","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Bloomberg GPT (few-shot, k=3)","paper":"/paper/bloomberggpt-a-large-language-model-for","metrics":{"Accuracy":"92"},"code_links":[{"title":"yangletliu/finlora","url":"https://github.com/yangletliu/finlora"},{"title":"open-finance-lab/finlora","url":"https://github.com/open-finance-lab/finlora"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-28","task":"Multiple Choice Question Answering (MCQA)","dataset_variant":"BIG-bench (Movie Recommendation)","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=3, CoT)","paper":"/paper/palm-2-technical-report-1","metrics":{"Accuracy":"94.4"},"code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-29","task":"Multiple Choice Question Answering (MCQA)","dataset_variant":"BIG-bench (Navigate)","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=3, CoT)","paper":"/paper/palm-2-technical-report-1","metrics":{"Accuracy":"91.2"},"code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-30","task":"Multiple Choice Question Answering (MCQA)","dataset_variant":"BIG-bench (Ruin Names)","rows":9,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2 (few-shot, k=3, Direct)","paper":"/paper/palm-2-technical-report-1","metrics":{"Accuracy":"90"},"code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/common-sense-reasoning-on-big-bench-sports","task":"Common Sense Reasoning","dataset_variant":"BIG-bench (Sports Understanding)","rows":8,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2(few-shot, k=3, CoT)","paper":"/paper/palm-2-technical-report-1","metrics":{"Accuracy":"98"},"code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/sarcasm-detection-on-big-bench-snarks","task":"Sarcasm Detection","dataset_variant":"BIG-bench (SNARKS)","rows":8,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM 2(few-shot, k=3, CoT)","paper":"/paper/palm-2-technical-report-1","metrics":{"Accuracy":"84.8"},"code_links":[{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multi-task-language-understanding-on-bbh-alg","task":"Multi-task Language Understanding","dataset_variant":"BBH-alg","rows":7,"metrics":["Average (%)"],"first_row_in_archive_order":{"model":"code-davinci-002 175B (CoT)","paper":"/paper/evaluating-large-language-models-trained-on","metrics":{"Average (%)":"73.9"},"code_links":[{"title":"THUDM/CodeGeeX","url":"https://github.com/THUDM/CodeGeeX"},{"title":"ncoop57/gpt-code-clippy","url":"https://github.com/ncoop57/gpt-code-clippy"},{"title":"codedotal/gpt-code-clippy","url":"https://github.com/codedotal/gpt-code-clippy"},{"title":"openai/human-eval","url":"https://github.com/openai/human-eval"},{"title":"vhellendoorn/code-lms","url":"https://github.com/vhellendoorn/code-lms"},{"title":"glouppe/info8010-deep-learning","url":"https://github.com/glouppe/info8010-deep-learning"},{"title":"microsoft/PythonProgrammingPuzzles","url":"https://github.com/microsoft/PythonProgrammingPuzzles"},{"title":"fsoft-ai4code/codecapybara","url":"https://github.com/fsoft-ai4code/codecapybara"},{"title":"codefuse-ai/codefuse-evaluation","url":"https://github.com/codefuse-ai/codefuse-evaluation"},{"title":"my-other-github-account/llm-humaneval-benchmarks","url":"https://github.com/my-other-github-account/llm-humaneval-benchmarks"},{"title":"superli3/codenavi","url":"https://github.com/superli3/codenavi"},{"title":"superli3/CYRMPR","url":"https://github.com/superli3/CYRMPR"},{"title":"2796gaurav/human-eval","url":"https://github.com/2796gaurav/human-eval"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/word-sense-disambiguation-on-big-bench","task":"Word Sense Disambiguation","dataset_variant":"BIG-bench (Anachronisms)","rows":6,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"69.1"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/common-sense-reasoning-on-big-bench-winowhy","task":"Common Sense Reasoning","dataset_variant":"BIG-bench (Winowhy)","rows":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM-540B (few-shot, k=5)","paper":"/paper/palm-scaling-language-modeling-with-pathways-1","metrics":{"Accuracy":"65.9"},"code_links":[{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"lucidrains/PaLM-pytorch","url":"https://github.com/lucidrains/PaLM-pytorch"},{"title":"google/paxml","url":"https://github.com/google/paxml"},{"title":"foundation-model-stack/fms-fsdp","url":"https://github.com/foundation-model-stack/fms-fsdp"},{"title":"lucidrains/PaLM-jax","url":"https://github.com/lucidrains/PaLM-jax"},{"title":"chrisociepa/allamo","url":"https://github.com/chrisociepa/allamo"},{"title":"conceptofmind/PaLM-flax","url":"https://github.com/conceptofmind/PaLM-flax"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/crass-ai-on-big-bench","task":"Crass AI","dataset_variant":"BIG-bench","rows":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Orca 2-13B","paper":"/paper/orca-2-teaching-small-language-models-how-to","metrics":{"Accuracy":"86.86"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/logical-reasoning-on-big-bench-logic-grid","task":"Logical Reasoning","dataset_variant":"BIG-bench (Logic Grid Puzzle)","rows":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"44"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/logical-reasoning-on-big-bench-strategyqa","task":"Logical Reasoning","dataset_variant":"BIG-bench (StrategyQA)","rows":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM-540B (few-shot, k=5)","paper":"/paper/palm-scaling-language-modeling-with-pathways-1","metrics":{"Accuracy":"73.9"},"code_links":[{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"lucidrains/PaLM-pytorch","url":"https://github.com/lucidrains/PaLM-pytorch"},{"title":"google/paxml","url":"https://github.com/google/paxml"},{"title":"foundation-model-stack/fms-fsdp","url":"https://github.com/foundation-model-stack/fms-fsdp"},{"title":"lucidrains/PaLM-jax","url":"https://github.com/lucidrains/PaLM-jax"},{"title":"chrisociepa/allamo","url":"https://github.com/chrisociepa/allamo"},{"title":"conceptofmind/PaLM-flax","url":"https://github.com/conceptofmind/PaLM-flax"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/multiple-choice-question-answering-mcqa-on-31","task":"Multiple Choice Question Answering (MCQA)","dataset_variant":"BIG-bench (Novel Concepts)","rows":4,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM-540B (few-shot, k=5)","paper":"/paper/palm-scaling-language-modeling-with-pathways-1","metrics":{"Accuracy":"71.9"},"code_links":[{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"lucidrains/PaLM-pytorch","url":"https://github.com/lucidrains/PaLM-pytorch"},{"title":"google/paxml","url":"https://github.com/google/paxml"},{"title":"foundation-model-stack/fms-fsdp","url":"https://github.com/foundation-model-stack/fms-fsdp"},{"title":"lucidrains/PaLM-jax","url":"https://github.com/lucidrains/PaLM-jax"},{"title":"chrisociepa/allamo","url":"https://github.com/chrisociepa/allamo"},{"title":"conceptofmind/PaLM-flax","url":"https://github.com/conceptofmind/PaLM-flax"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/auto-debugging-on-big-bench-lite","task":"Auto Debugging","dataset_variant":"Big-bench Lite","rows":3,"metrics":["Exact string match"],"first_row_in_archive_order":{"model":"PaLM 62B (few-shot, k=5)","paper":"/paper/palm-scaling-language-modeling-with-pathways-1","metrics":{"Exact string match":"38.2"},"code_links":[{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"lucidrains/PaLM-pytorch","url":"https://github.com/lucidrains/PaLM-pytorch"},{"title":"google/paxml","url":"https://github.com/google/paxml"},{"title":"foundation-model-stack/fms-fsdp","url":"https://github.com/foundation-model-stack/fms-fsdp"},{"title":"lucidrains/PaLM-jax","url":"https://github.com/lucidrains/PaLM-jax"},{"title":"chrisociepa/allamo","url":"https://github.com/chrisociepa/allamo"},{"title":"conceptofmind/PaLM-flax","url":"https://github.com/conceptofmind/PaLM-flax"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/common-sense-reasoning-on-big-bench-known","task":"Common Sense Reasoning","dataset_variant":"BIG-bench (Known Unknowns)","rows":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM-540B (few-shot, k=5)","paper":"/paper/palm-scaling-language-modeling-with-pathways-1","metrics":{"Accuracy":"73.9"},"code_links":[{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"lucidrains/PaLM-pytorch","url":"https://github.com/lucidrains/PaLM-pytorch"},{"title":"google/paxml","url":"https://github.com/google/paxml"},{"title":"foundation-model-stack/fms-fsdp","url":"https://github.com/foundation-model-stack/fms-fsdp"},{"title":"lucidrains/PaLM-jax","url":"https://github.com/lucidrains/PaLM-jax"},{"title":"chrisociepa/allamo","url":"https://github.com/chrisociepa/allamo"},{"title":"conceptofmind/PaLM-flax","url":"https://github.com/conceptofmind/PaLM-flax"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/language-modelling-on-big-bench-lite","task":"Language Modelling","dataset_variant":"BIG-bench-lite","rows":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"GLM-130B (3-shot)","paper":"/paper/glm-130b-an-open-bilingual-pre-trained-model","metrics":{"Accuracy":"15.11"},"code_links":[{"title":"thudm/chatglm2-6b","url":"https://github.com/thudm/chatglm2-6b"},{"title":"thudm/chatglm3","url":"https://github.com/thudm/chatglm3"},{"title":"thudm/chatglm","url":"https://github.com/thudm/chatglm"},{"title":"modelscope/modelscope","url":"https://github.com/modelscope/modelscope"},{"title":"thudm/glm-130b","url":"https://github.com/thudm/glm-130b"},{"title":"THUDM/GLM","url":"https://github.com/THUDM/GLM"},{"title":"jackaduma/ChatGLM-LoRA-RLHF-PyTorch","url":"https://github.com/jackaduma/ChatGLM-LoRA-RLHF-PyTorch"},{"title":"2023-MindSpore-4/Code12","url":"https://github.com/2023-MindSpore-4/Code12/tree/main/MindFormers/glm"},{"title":"2023-MindSpore-4/Code12","url":"https://github.com/2023-MindSpore-4/Code12/tree/main/MindFormers/glm3"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/memorization-on-big-bench-hindu-knowledge","task":"Memorization","dataset_variant":"BIG-bench (Hindu Knowledge)","rows":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM-540B (few-shot, k=5)","paper":"/paper/palm-scaling-language-modeling-with-pathways-1","metrics":{"Accuracy":"95.4"},"code_links":[{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"lucidrains/PaLM-pytorch","url":"https://github.com/lucidrains/PaLM-pytorch"},{"title":"google/paxml","url":"https://github.com/google/paxml"},{"title":"foundation-model-stack/fms-fsdp","url":"https://github.com/foundation-model-stack/fms-fsdp"},{"title":"lucidrains/PaLM-jax","url":"https://github.com/lucidrains/PaLM-jax"},{"title":"chrisociepa/allamo","url":"https://github.com/chrisociepa/allamo"},{"title":"conceptofmind/PaLM-flax","url":"https://github.com/conceptofmind/PaLM-flax"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/analogical-similarity-on-big-bench","task":"Analogical Similarity","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"38.1"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/analytic-entailment-on-big-bench","task":"Analytic Entailment","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"67.1"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/common-sense-reasoning-on-big-bench-logical","task":"Common Sense Reasoning","dataset_variant":"BIG-bench (Logical Sequence)","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"64.1"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/crash-blossom-on-big-bench","task":"Crash Blossom","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"63.6"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/dark-humor-detection-on-big-bench","task":"Dark Humor Detection","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"83.1"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/discourse-marker-prediction-on-big-bench","task":"Discourse Marker Prediction","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"13.1"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/empirical-judgments-on-big-bench","task":"Empirical Judgments","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"67.7"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/english-proverbs-on-big-bench","task":"English Proverbs","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy ":"82.4"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/entailed-polarity-on-big-bench","task":"Entailed Polarity","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"94"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/epistemic-reasoning-on-big-bench","task":"Epistemic Reasoning","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"60.6"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/evaluating-information-essentiality-on-big","task":"Evaluating Information Essentiality","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"17.6"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/fantasy-reasoning-on-big-bench","task":"Fantasy Reasoning","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy ":"69"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/figure-of-speech-detection-on-big-bench","task":"Figure Of Speech Detection","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy ":"63.3"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/general-knowledge-on-big-bench","task":"General Knowledge","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"94.3"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/gre-reading-comprehension-on-big-bench","task":"GRE Reading Comprehension","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy ":"53.1"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/human-organs-senses-multiple-choice-on-big","task":"Human Organs Senses Multiple Choice","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy ":"85.7"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/identify-odd-metapor-on-big-bench","task":"Identify Odd Metapor","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"68.8"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/implicatures-on-big-bench","task":"Implicatures","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"75"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/implicit-relations-on-big-bench","task":"Implicit Relations","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy ":"49.4"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/intent-recognition-on-big-bench","task":"Intent Recognition","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy ":"92.8"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/irony-identification-on-big-bench","task":"Irony Identification","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"73.0"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/lambada-on-big-bench","task":"LAMBADA","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Chinchilla-70B (zero-shot)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy ":"77.4"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/logical-args-on-big-bench","task":"Logical Args","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"59.1"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/logical-reasoning-on-big-bench-logical","task":"Logical Reasoning","dataset_variant":"BIG-bench (Logical Fallacy Detection)","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"72.1"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/mathematical-induction-on-big-bench","task":"Mathematical Induction","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"57.6"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/metaphor-boolean-on-big-bench","task":"Metaphor Boolean","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"93.1"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/misconceptions-on-big-bench","task":"Misconceptions","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"65.3"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/moral-permissibility-on-big-bench","task":"Moral Permissibility","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"57.3"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/movie-dialog-same-or-different-on-big-bench","task":"Movie Dialog Same Or Different","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy ":"54.5"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/nonsense-words-grammar-on-big-bench","task":"Nonsense Words Grammar","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"78"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/odd-one-out-on-big-bench","task":"Odd One Out","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"70.9"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/phrase-relatedness-on-big-bench","task":"Phrase Relatedness","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy ":"94"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/physical-intuition-on-big-bench","task":"Physical Intuition","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"79"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/presuppositions-as-nli-on-big-bench","task":"Presuppositions As NLI","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"49.9"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/question-selection-on-big-bench","task":"Question Selection","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"52.6"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/riddle-sense-on-big-bench","task":"Riddle Sense","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"85.7"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/sentence-ambiguity-on-big-bench","task":"Sentence Ambiguity","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"71.7"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/similarities-abstraction-on-big-bench","task":"Similarities Abstraction","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"87"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/timedial-on-big-bench","task":"Timedial","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy":"68.8"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/understanding-fables-on-big-bench","task":"Understanding Fables","dataset_variant":"BIG-bench","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Chinchilla-70B (few-shot, k=5)","paper":"/paper/training-compute-optimal-large-language","metrics":{"Accuracy ":"60.3"},"code_links":[{"title":"karpathy/llama2.c","url":"https://github.com/karpathy/llama2.c"},{"title":"nkluge-correa/teenytinyllama","url":"https://github.com/nkluge-correa/teenytinyllama"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/abstract-algebra-on-big-bench","task":"Abstract Algebra","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"25.0"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/anatomy-on-big-bench","task":"Anatomy","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"56.3"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/astronomy-on-big-bench","task":"Astronomy","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"65.8"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/business-ethics-on-big-bench","task":"Business Ethics","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"70.0"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/clinical-knowledge-on-big-bench","task":"Clinical Knowledge","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"67.2"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/college-mathematics-on-big-bench","task":"College Mathematics","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"37.0"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/college-medicine-on-big-bench","task":"College Medicine","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"60.1"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/computer-security-on-big-bench","task":"Computer Security","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"65.0"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/econometrics-on-big-bench","task":"Econometrics","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"43"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/elementary-mathematics-on-big-bench","task":"Elementary Mathematics","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"33.6"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/fever-2-way-on-big-bench","task":"FEVER (2-way)","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=10)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"77.5"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/fever-3-way-on-big-bench","task":"FEVER (3-way)","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=15)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"77.5"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/formal-logic-on-big-bench","task":"Formal Logic","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"35.7"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/global-facts-on-big-bench","task":"Global Facts","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"38.0"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/high-school-european-history-on-big-bench","task":"High School European History","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"72.1"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/high-school-geography-on-big-bench","task":"High School Geography","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"76.8"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/high-school-government-and-politics-on-big","task":"High School Government and Politics","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"83.9"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/high-school-macroeconomics-on-big-bench","task":"High School Macroeconomics","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"65.1"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/high-school-mathematics-on-big-bench","task":"High School Mathematics","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"23.7"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/high-school-microeconomics-on-big-bench","task":"High School Microeconomics","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"66.4"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/high-school-psychology-on-big-bench","task":"High School Psychology","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"81.8"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/high-school-us-history-on-big-bench","task":"High School US History","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"78.9"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/high-school-world-history-on-big-bench","task":"High School World History","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"75.1"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/human-aging-on-big-bench","task":"Human Aging","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"66.4"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/human-sexuality-on-big-bench","task":"Human Sexuality","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"67.2"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/international-law-on-big-bench","task":"International Law","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"77.7"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/jurisprudence-on-big-bench","task":"Jurisprudence","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"71.3"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/logical-fallacies-on-big-bench","task":"Logical Fallacies","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"72.4"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/machine-learning-on-big-bench","task":"BIG-bench Machine Learning","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"41.1"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/management-on-big-bench","task":"Management","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"77.7"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/marketing-on-big-bench","task":"Marketing","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"83.3"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/medical-genetics-on-big-bench","task":"Medical Genetics","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"69.0"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/miscellaneous-on-big-bench","task":"Miscellaneous","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"75.7"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/moral-disputes-on-big-bench","task":"Moral Disputes","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"66.8"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/moral-scenarios-on-big-bench","task":"Moral Scenarios","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"40.2"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/natural-questions-on-big-bench","task":"Natural Questions","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=64)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"28.2"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/nutrition-on-big-bench","task":"Nutrition","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"69.9"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/on-big-bench-hyperbaton","task":"","dataset_variant":"BIG-bench (Hyperbaton)","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"CoT-T5 11B","paper":"/paper/the-cot-collection-improving-zero-shot-and","metrics":{"Accuracy":"65.2"},"code_links":[{"title":"kaistai/cot-collection","url":"https://github.com/kaistai/cot-collection"},{"title":"kaist-lklab/cot-collection","url":"https://github.com/kaist-lklab/cot-collection"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/on-big-bench-navigate","task":"","dataset_variant":"BIG-bench (Navigate)","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"CoT-T5 11B","paper":"/paper/the-cot-collection-improving-zero-shot-and","metrics":{"Accuracy":"60"},"code_links":[{"title":"kaistai/cot-collection","url":"https://github.com/kaistai/cot-collection"},{"title":"kaist-lklab/cot-collection","url":"https://github.com/kaist-lklab/cot-collection"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/on-big-bench-ruin-names","task":"","dataset_variant":"BIG-bench (Ruin Names)","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"CoT-T5 11B","paper":"/paper/the-cot-collection-improving-zero-shot-and","metrics":{"Accuracy":"42.8"},"code_links":[{"title":"kaistai/cot-collection","url":"https://github.com/kaistai/cot-collection"},{"title":"kaist-lklab/cot-collection","url":"https://github.com/kaist-lklab/cot-collection"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/on-big-bench-snarks","task":"","dataset_variant":"BIG-bench (SNARKS)","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"CoT-T5 11B","paper":"/paper/the-cot-collection-improving-zero-shot-and","metrics":{"Accuracy":"67.7"},"code_links":[{"title":"kaistai/cot-collection","url":"https://github.com/kaistai/cot-collection"},{"title":"kaist-lklab/cot-collection","url":"https://github.com/kaist-lklab/cot-collection"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/philosophy-on-big-bench","task":"Philosophy","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"68.8"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/prehistory-on-big-bench","task":"Prehistory","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"67.6"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/professional-accounting-on-big-bench","task":"Professional Accounting","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"44.3"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/professional-law-on-big-bench","task":"Professional Law","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"44.5"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/professional-medicine-on-big-bench","task":"Professional Medicine","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"64.0"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/professional-psychology-on-big-bench","task":"Professional Psychology","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"68.1"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/public-relations-on-big-bench","task":"Public Relations","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"71.8"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/race-h-on-big-bench","task":"RACE-h","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"71.6"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/race-m-on-big-bench","task":"RACE-m","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"75.1"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/security-studies-on-big-bench","task":"Security Studies","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"64.9"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/sociology-on-big-bench","task":"Sociology","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"84.1"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/triviaqa-on-big-bench","task":"TriviaQA","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=64)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"57.1"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/us-foreign-policy-on-big-bench","task":"US Foreign Policy","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy ":"81.0"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/virology-on-big-bench","task":"Virology","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"47.0"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/world-religions-on-big-bench","task":"World Religions","dataset_variant":"BIG-bench","rows":1,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"Gopher-280B (few-shot, k=5)","paper":"/paper/scaling-language-models-methods-analysis-1","metrics":{"Accuracy":"84.2"},"code_links":[{"title":"allenai/dolma","url":"https://github.com/allenai/dolma"},{"title":"rvlopes/gloria","url":"https://github.com/rvlopes/gloria"},{"title":"bramiozo/PubScience","url":"https://github.com/bramiozo/PubScience"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/orca-2-teaching-small-language-models-how-to","title":"Orca 2: Teaching Small Language Models How to Reason","date":"2023-11-18","rows_on_this_dataset":4,"code_links":0,"syntology":null},{"paper":"/paper/the-cot-collection-improving-zero-shot-and","title":"The CoT Collection: Improving Zero-shot and Few-shot Learning of Language Models via Chain-of-Thought Fine-Tuning","date":"2023-05-23","rows_on_this_dataset":4,"code_links":2,"syntology":null},{"paper":"/paper/palm-2-technical-report-1","title":"PaLM 2 Technical Report","date":"2023-05-17","rows_on_this_dataset":28,"code_links":1,"syntology":null},{"paper":"/paper/bloomberggpt-a-large-language-model-for","title":"BloombergGPT: A Large Language Model for Finance","date":"2023-03-30","rows_on_this_dataset":63,"code_links":2,"syntology":null},{"paper":"/paper/galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":2,"samples_ran":0,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scaling-instruction-finetuned-language-models","title":"Scaling Instruction-Finetuned Language Models","date":"2022-10-20","rows_on_this_dataset":12,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":17,"samples_ran":8,"samples_unverified":9,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/glm-130b-an-open-bilingual-pre-trained-model","title":"GLM-130B: An Open Bilingual Pre-trained Model","date":"2022-10-05","rows_on_this_dataset":3,"code_links":9,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":21,"samples_ran":5,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","rows_on_this_dataset":12,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":37,"samples_ran":30,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/training-compute-optimal-large-language","title":"Training Compute-Optimal Large Language Models","date":"2022-03-29","rows_on_this_dataset":60,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":8,"samples_unverified":3,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/scaling-language-models-methods-analysis-1","title":"Scaling Language Models: Methods, Analysis & Insights from Training Gopher","date":"2021-12-08","rows_on_this_dataset":113,"code_links":3,"syntology":null},{"paper":"/paper/evaluating-large-language-models-trained-on","title":"Evaluating Large Language Models Trained on Code","date":"2021-07-07","rows_on_this_dataset":2,"code_links":13,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":39,"samples_ran":6,"samples_unverified":33,"pointer_only_for_licence":2,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":6,"samples_harvested":127,"samples_ran":57,"samples_unverified":70,"pointer_only_for_licence":8,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}