{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multi-task-language-understanding/papers/ran/1","list_of":"/task/multi-task-language-understanding","task":"Multi-task Language Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,27],"of":27,"counts":{"archive_papers_tagged":57,"with_a_code_link":44,"where_syntology_ran_a_sample":27,"not_listed_spam_title":0,"listed":57,"listed_where_code_ran":27,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":27,"every_run_a_failure_of_syntologys_instrument":0,"listed_with_a_run_with_no_instrument_failure":27,"listed_every_run_a_failure_of_syntologys_instrument":0,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multi-task-language-understanding/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/the-llama-3-herd-of-models","slug":"the-llama-3-herd-of-models","title":"The Llama 3 Herd of Models","date":"2024-07-31","arxiv_id":"2407.21783","repositories_listed":5,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-llama-3-herd-of-models#ran","syntology_url":"https://syntology.ai/paper/2407.21783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21783"}},"official":null}},{"url":"/paper/breaking-the-ceiling-of-the-llm-community-by","slug":"breaking-the-ceiling-of-the-llm-community-by","title":"Breaking the Ceiling of the LLM Community by Treating Token Generation as a Classification for Ensembling","date":"2024-06-18","arxiv_id":"2406.12585","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/breaking-the-ceiling-of-the-llm-community-by#ran","syntology_url":"https://syntology.ai/paper/2406.12585","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12585"}},"official":{"repos":["yaoching0/gac"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mmlu-pro-a-more-robust-and-challenging-multi","slug":"mmlu-pro-a-more-robust-and-challenging-multi","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","date":"2024-06-03","arxiv_id":"2406.01574","repositories_listed":2,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":3,"n_instrument":9,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 9 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mmlu-pro-a-more-robust-and-challenging-multi#ran","syntology_url":"https://syntology.ai/paper/2406.01574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01574"}},"official":null}},{"url":"/paper/arabicmmlu-assessing-massive-multitask","slug":"arabicmmlu-assessing-massive-multitask","title":"ArabicMMLU: Assessing Massive Multitask Language Understanding in Arabic","date":"2024-02-20","arxiv_id":"2402.12840","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/arabicmmlu-assessing-massive-multitask#ran","syntology_url":"https://syntology.ai/paper/2402.12840","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12840"}},"official":{"repos":["mbzuai-nlp/arabicmmlu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/leeroo-orchestrator-elevating-llms","slug":"leeroo-orchestrator-elevating-llms","title":"Routoo: Learning to Route to Large Language Models Effectively","date":"2024-01-25","arxiv_id":"2401.13979","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leeroo-orchestrator-elevating-llms#ran","syntology_url":"https://syntology.ai/paper/2401.13979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13979"}},"official":{"repos":["leeroo-ai/leeroo_orchestrator"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mixtral-of-experts","slug":"mixtral-of-experts","title":"Mixtral of Experts","date":"2024-01-08","arxiv_id":"2401.04088","repositories_listed":6,"syntology":{"n":5,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/mixtral-of-experts#ran","syntology_url":"https://syntology.ai/paper/2401.04088","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04088"}},"official":null}},{"url":"/paper/mistral-7b","slug":"mistral-7b","title":"Mistral 7B","date":"2023-10-10","arxiv_id":"2310.06825","repositories_listed":6,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mistral-7b#ran","syntology_url":"https://syntology.ai/paper/2310.06825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06825"}},"official":{"repos":["mistralai/mistral-src"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/large-language-models-only-pass-primary","slug":"large-language-models-only-pass-primary","title":"Large Language Models Only Pass Primary School Exams in Indonesia: A Comprehensive Test on IndoMMLU","date":"2023-10-07","arxiv_id":"2310.04928","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/large-language-models-only-pass-primary#ran","syntology_url":"https://syntology.ai/paper/2310.04928","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04928"}},"official":{"repos":["fajri91/indommlu"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/are-human-generated-demonstrations-necessary","slug":"are-human-generated-demonstrations-necessary","title":"Are Human-generated Demonstrations Necessary for In-context Learning?","date":"2023-09-26","arxiv_id":"2309.14681","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/are-human-generated-demonstrations-necessary#ran","syntology_url":"https://syntology.ai/paper/2309.14681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.14681"}},"official":{"repos":["ruili33/sec"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llama-2-open-foundation-and-fine-tuned-chat","slug":"llama-2-open-foundation-and-fine-tuned-chat","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","date":"2023-07-18","arxiv_id":"2307.09288","repositories_listed":19,"syntology":{"n":52,"n_ran":33,"n_constructed":7,"n_ran_checked":22,"n_instrument":11,"n_unverified":19,"n_honours":1,"n_violates":1,"n_no_contract":20,"n_pointer_only":20,"phrase":"33 ran (of which 7 constructed an object rather than computing a result; 22 with no instrument failure: 1 honoured, 1 violated, 20 with no contract checked; 11 where Syntology's instrument failed) · 19 unverified","sample_list":"/paper/llama-2-open-foundation-and-fine-tuned-chat#ran","syntology_url":"https://syntology.ai/paper/2307.09288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.09288"}},"official":{"repos":["facebookresearch/llama"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/gpt-4-technical-report-1","slug":"gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","arxiv_id":"2303.08774","repositories_listed":11,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpt-4-technical-report-1#ran","syntology_url":"https://syntology.ai/paper/2303.08774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08774"}},"official":{"repos":["openai/evals"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/llama-open-and-efficient-foundation-language-1","slug":"llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","arxiv_id":"2302.13971","repositories_listed":57,"syntology":{"n":58,"n_ran":37,"n_constructed":9,"n_ran_checked":25,"n_instrument":12,"n_unverified":21,"n_honours":3,"n_violates":0,"n_no_contract":22,"n_pointer_only":4,"phrase":"37 ran (of which 9 constructed an object rather than computing a result; 25 with no instrument failure: 3 honoured, 0 violated, 22 with no contract checked; 12 where Syntology's instrument failed) · 21 unverified","sample_list":"/paper/llama-open-and-efficient-foundation-language-1#ran","syntology_url":"https://syntology.ai/paper/2302.13971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.13971"}},"official":{"repos":["facebookresearch/llama"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/replug-retrieval-augmented-black-box-language","slug":"replug-retrieval-augmented-black-box-language","title":"REPLUG: Retrieval-Augmented Black-Box Language Models","date":"2023-01-30","arxiv_id":"2301.12652","repositories_listed":3,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/replug-retrieval-augmented-black-box-language#ran","syntology_url":"https://syntology.ai/paper/2301.12652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12652"}},"official":null}},{"url":"/paper/galactica-a-large-language-model-for-science-1","slug":"galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","arxiv_id":"2211.09085","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/galactica-a-large-language-model-for-science-1#ran","syntology_url":"https://syntology.ai/paper/2211.09085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09085"}},"official":{"repos":["paperswithcode/galai"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-instruction-finetuned-language-models","slug":"scaling-instruction-finetuned-language-models","title":"Scaling Instruction-Finetuned Language Models","date":"2022-10-20","arxiv_id":"2210.11416","repositories_listed":9,"syntology":{"n":17,"n_ran":8,"n_constructed":1,"n_ran_checked":1,"n_instrument":7,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 7 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/scaling-instruction-finetuned-language-models#ran","syntology_url":"https://syntology.ai/paper/2210.11416","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11416"}},"official":null}},{"url":"/paper/glm-130b-an-open-bilingual-pre-trained-model","slug":"glm-130b-an-open-bilingual-pre-trained-model","title":"GLM-130B: An Open Bilingual Pre-trained Model","date":"2022-10-05","arxiv_id":"2210.02414","repositories_listed":9,"syntology":{"n":21,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/glm-130b-an-open-bilingual-pre-trained-model#ran","syntology_url":"https://syntology.ai/paper/2210.02414","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.02414"}},"official":{"repos":["thudm/glm-130b"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/few-shot-learning-with-retrieval-augmented","slug":"few-shot-learning-with-retrieval-augmented","title":"Atlas: Few-shot Learning with Retrieval Augmented Language Models","date":"2022-08-05","arxiv_id":"2208.03299","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/few-shot-learning-with-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2208.03299","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.03299"}},"official":{"repos":["facebookresearch/atlas"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-language-learning-paradigms","slug":"unifying-language-learning-paradigms","title":"UL2: Unifying Language Learning Paradigms","date":"2022-05-10","arxiv_id":"2205.05131","repositories_listed":2,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unifying-language-learning-paradigms#ran","syntology_url":"https://syntology.ai/paper/2205.05131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05131"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/gpt-neox-20b-an-open-source-autoregressive-1","slug":"gpt-neox-20b-an-open-source-autoregressive-1","title":"GPT-NeoX-20B: An Open-Source Autoregressive Language Model","date":"2022-04-14","arxiv_id":"2204.06745","repositories_listed":11,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gpt-neox-20b-an-open-source-autoregressive-1#ran","syntology_url":"https://syntology.ai/paper/2204.06745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.06745"}},"official":{"repos":["eleutherai/gpt-neox"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/palm-scaling-language-modeling-with-pathways-1","slug":"palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","arxiv_id":"2204.02311","repositories_listed":7,"syntology":{"n":37,"n_ran":32,"n_constructed":16,"n_ran_checked":24,"n_instrument":8,"n_unverified":5,"n_honours":2,"n_violates":1,"n_no_contract":21,"n_pointer_only":0,"phrase":"32 ran (of which 16 constructed an object rather than computing a result; 24 with no instrument failure: 2 honoured, 1 violated, 21 with no contract checked; 8 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/palm-scaling-language-modeling-with-pathways-1#ran","syntology_url":"https://syntology.ai/paper/2204.02311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02311"}},"official":null}},{"url":"/paper/training-compute-optimal-large-language","slug":"training-compute-optimal-large-language","title":"Training Compute-Optimal Large Language Models","date":"2022-03-29","arxiv_id":"2203.15556","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":3,"n_ran_checked":5,"n_instrument":3,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/training-compute-optimal-large-language#ran","syntology_url":"https://syntology.ai/paper/2203.15556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15556"}},"official":null}},{"url":"/paper/evaluating-large-language-models-trained-on","slug":"evaluating-large-language-models-trained-on","title":"Evaluating Large Language Models Trained on Code","date":"2021-07-07","arxiv_id":"2107.03374","repositories_listed":13,"syntology":{"n":39,"n_ran":26,"n_constructed":0,"n_ran_checked":24,"n_instrument":2,"n_unverified":13,"n_honours":1,"n_violates":0,"n_no_contract":23,"n_pointer_only":4,"phrase":"26 ran (of which 0 constructed an object rather than computing a result; 24 with no instrument failure: 1 honoured, 0 violated, 23 with no contract checked; 2 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/evaluating-large-language-models-trained-on#ran","syntology_url":"https://syntology.ai/paper/2107.03374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.03374"}},"official":{"repos":["openai/human-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["found_in_text","listed","official"]}}},{"url":"/paper/measuring-massive-multitask-language","slug":"measuring-massive-multitask-language","title":"Measuring Massive Multitask Language Understanding","date":"2020-09-07","arxiv_id":"2009.03300","repositories_listed":18,"syntology":{"n":26,"n_ran":19,"n_constructed":0,"n_ran_checked":15,"n_instrument":4,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":1,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/measuring-massive-multitask-language#ran","syntology_url":"https://syntology.ai/paper/2009.03300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.03300"}},"official":{"repos":["hendrycks/test"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/language-models-are-few-shot-learners","slug":"language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","arxiv_id":"2005.14165","repositories_listed":67,"syntology":{"n":65,"n_ran":45,"n_constructed":0,"n_ran_checked":40,"n_instrument":5,"n_unverified":20,"n_honours":2,"n_violates":1,"n_no_contract":37,"n_pointer_only":7,"phrase":"45 ran (of which 0 constructed an object rather than computing a result; 40 with no instrument failure: 2 honoured, 1 violated, 37 with no contract checked; 5 where Syntology's instrument failed) · 20 unverified","sample_list":"/paper/language-models-are-few-shot-learners#ran","syntology_url":"https://syntology.ai/paper/2005.14165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.14165"}},"official":{"repos":["openai/gpt-3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/unifiedqa-crossing-format-boundaries-with-a","slug":"unifiedqa-crossing-format-boundaries-with-a","title":"UnifiedQA: Crossing Format Boundaries With a Single QA System","date":"2020-05-02","arxiv_id":"2005.00700","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unifiedqa-crossing-format-boundaries-with-a#ran","syntology_url":"https://syntology.ai/paper/2005.00700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.00700"}},"official":{"repos":["allenai/unifiedqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/albert-a-lite-bert-for-self-supervised","slug":"albert-a-lite-bert-for-self-supervised","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","date":"2019-09-26","arxiv_id":"1909.11942","repositories_listed":48,"syntology":{"n":126,"n_ran":81,"n_constructed":17,"n_ran_checked":59,"n_instrument":22,"n_unverified":45,"n_honours":4,"n_violates":0,"n_no_contract":55,"n_pointer_only":28,"phrase":"81 ran (of which 17 constructed an object rather than computing a result; 59 with no instrument failure: 4 honoured, 0 violated, 55 with no contract checked; 22 where Syntology's instrument failed) · 45 unverified","sample_list":"/paper/albert-a-lite-bert-for-self-supervised#ran","syntology_url":"https://syntology.ai/paper/1909.11942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.11942"}},"official":{"repos":["google-research/ALBERT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/roberta-a-robustly-optimized-bert-pretraining","slug":"roberta-a-robustly-optimized-bert-pretraining","title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach","date":"2019-07-26","arxiv_id":"1907.11692","repositories_listed":67,"syntology":{"n":48,"n_ran":37,"n_constructed":11,"n_ran_checked":36,"n_instrument":1,"n_unverified":11,"n_honours":0,"n_violates":0,"n_no_contract":36,"n_pointer_only":24,"phrase":"37 ran (of which 11 constructed an object rather than computing a result; 36 with no instrument failure: 0 honoured, 0 violated, 36 with no contract checked; 1 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/roberta-a-robustly-optimized-bert-pretraining#ran","syntology_url":"https://syntology.ai/paper/1907.11692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.11692"}},"official":{"repos":["pytorch/fairseq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}}],"record_sha256":"6c4c7005a15c8808d52ae681ab789557717224687191356cc13bbf774cdf2954","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}