{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mmlu/papers/ran/1","list_of":"/task/mmlu","task":"MMLU","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,78],"of":78,"counts":{"archive_papers_tagged":340,"with_a_code_link":156,"where_syntology_ran_a_sample":78,"not_listed_spam_title":0,"listed":340,"listed_where_code_ran":78,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":60,"every_run_a_failure_of_syntologys_instrument":18,"listed_with_a_run_with_no_instrument_failure":60,"listed_every_run_a_failure_of_syntologys_instrument":18,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mmlu/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/any4-learned-4-bit-numeric-representation-for","slug":"any4-learned-4-bit-numeric-representation-for","title":"any4: Learned 4-bit Numeric Representation for LLMs","date":"2025-07-07","arxiv_id":"2507.04610","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/any4-learned-4-bit-numeric-representation-for#ran","syntology_url":"https://syntology.ai/paper/2507.04610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.04610"}},"official":{"repos":["facebookresearch/any4"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-fine-tuning-naturally-mitigates","slug":"reinforcement-fine-tuning-naturally-mitigates","title":"Reinforcement Fine-Tuning Naturally Mitigates Forgetting in Continual Post-Training","date":"2025-07-07","arxiv_id":"2507.05386","repositories_listed":0,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reinforcement-fine-tuning-naturally-mitigates#ran","syntology_url":"https://syntology.ai/paper/2507.05386","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.05386"}},"official":null}},{"url":"/paper/helm-hyperbolic-large-language-models-via","slug":"helm-hyperbolic-large-language-models-via","title":"HELM: Hyperbolic Large Language Models via Mixture-of-Curvature Experts","date":"2025-05-30","arxiv_id":"2505.24722","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/helm-hyperbolic-large-language-models-via#ran","syntology_url":"https://syntology.ai/paper/2505.24722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24722"}},"official":{"repos":["graph-and-geometric-learning/helm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/actor-critic-based-online-data-mixing-for","slug":"actor-critic-based-online-data-mixing-for","title":"Actor-Critic based Online Data Mixing For Language Model Pre-Training","date":"2025-05-29","arxiv_id":"2505.23878","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/actor-critic-based-online-data-mixing-for#ran","syntology_url":"https://syntology.ai/paper/2505.23878","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23878"}},"official":null}},{"url":"/paper/reinforcing-general-reasoning-without","slug":"reinforcing-general-reasoning-without","title":"Reinforcing General Reasoning without Verifiers","date":"2025-05-27","arxiv_id":"2505.21493","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcing-general-reasoning-without#ran","syntology_url":"https://syntology.ai/paper/2505.21493","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21493"}},"official":{"repos":["sail-sg/verifree"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/capability-based-scaling-laws-for-llm-red","slug":"capability-based-scaling-laws-for-llm-red","title":"Capability-Based Scaling Laws for LLM Red-Teaming","date":"2025-05-26","arxiv_id":"2505.20162","repositories_listed":1,"syntology":{"n":23,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":0,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/capability-based-scaling-laws-for-llm-red#ran","syntology_url":"https://syntology.ai/paper/2505.20162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20162"}},"official":{"repos":["kotekjedi/capability-based-scaling"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":17,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/reasonir-training-retrievers-for-reasoning","slug":"reasonir-training-retrievers-for-reasoning","title":"ReasonIR: Training Retrievers for Reasoning Tasks","date":"2025-04-29","arxiv_id":"2504.20595","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reasonir-training-retrievers-for-reasoning#ran","syntology_url":"https://syntology.ai/paper/2504.20595","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20595"}},"official":{"repos":["facebookresearch/reasonir"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/datadecide-how-to-predict-best-pretraining","slug":"datadecide-how-to-predict-best-pretraining","title":"DataDecide: How to Predict Best Pretraining Data with Small Experiments","date":"2025-04-15","arxiv_id":"2504.11393","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/datadecide-how-to-predict-best-pretraining#ran","syntology_url":"https://syntology.ai/paper/2504.11393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11393"}},"official":null}},{"url":"/paper/dynamic-cheatsheet-test-time-learning-with","slug":"dynamic-cheatsheet-test-time-learning-with","title":"Dynamic Cheatsheet: Test-Time Learning with Adaptive Memory","date":"2025-04-10","arxiv_id":"2504.07952","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-cheatsheet-test-time-learning-with#ran","syntology_url":"https://syntology.ai/paper/2504.07952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07952"}},"official":{"repos":["suzgunmirac/dynamic-cheatsheet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/right-question-is-already-half-the-answer","slug":"right-question-is-already-half-the-answer","title":"Right Question is Already Half the Answer: Fully Unsupervised LLM Reasoning Incentivization","date":"2025-04-08","arxiv_id":"2504.05812","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/right-question-is-already-half-the-answer#ran","syntology_url":"https://syntology.ai/paper/2504.05812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05812"}},"official":{"repos":["qingyangzhang/empo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/effective-skill-unlearning-through","slug":"effective-skill-unlearning-through","title":"Effective Skill Unlearning through Intervention and Abstention","date":"2025-03-27","arxiv_id":"2503.21730","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/effective-skill-unlearning-through#ran","syntology_url":"https://syntology.ai/paper/2503.21730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21730"}},"official":{"repos":["trustworthy-ml-lab/effective_skill_unlearning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chatbench-from-static-benchmarks-to-human-ai","slug":"chatbench-from-static-benchmarks-to-human-ai","title":"ChatBench: From Static Benchmarks to Human-AI Evaluation","date":"2025-03-22","arxiv_id":"2504.07114","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chatbench-from-static-benchmarks-to-human-ai#ran","syntology_url":"https://syntology.ai/paper/2504.07114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07114"}},"official":{"repos":["serinachang5/interactive-eval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/score-systematic-consistency-and-robustness","slug":"score-systematic-consistency-and-robustness","title":"SCORE: Systematic COnsistency and Robustness Evaluation for Large Language Models","date":"2025-02-28","arxiv_id":"2503.00137","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/score-systematic-consistency-and-robustness#ran","syntology_url":"https://syntology.ai/paper/2503.00137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00137"}},"official":{"repos":["EleutherAI/lm-evaluation-harness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/earlier-tokens-contribute-more-learning","slug":"earlier-tokens-contribute-more-learning","title":"Earlier Tokens Contribute More: Learning Direct Preference Optimization From Temporal Decay Perspective","date":"2025-02-20","arxiv_id":"2502.14340","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/earlier-tokens-contribute-more-learning#ran","syntology_url":"https://syntology.ai/paper/2502.14340","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14340"}},"official":{"repos":["lotusrc/d2po"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/noise-injection-reveals-hidden-capabilities","slug":"noise-injection-reveals-hidden-capabilities","title":"Noise Injection Reveals Hidden Capabilities of Sandbagging Language Models","date":"2024-12-02","arxiv_id":"2412.01784","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/noise-injection-reveals-hidden-capabilities#ran","syntology_url":"https://syntology.ai/paper/2412.01784","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.01784"}},"official":{"repos":["camtice/sandbagdetect"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/todo-enhancing-llm-alignment-with-ternary","slug":"todo-enhancing-llm-alignment-with-ternary","title":"TODO: Enhancing LLM Alignment with Ternary Preferences","date":"2024-11-02","arxiv_id":"2411.02442","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/todo-enhancing-llm-alignment-with-ternary#ran","syntology_url":"https://syntology.ai/paper/2411.02442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02442"}},"official":{"repos":["xxares/todo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/shopping-mmlu-a-massive-multi-task-online","slug":"shopping-mmlu-a-massive-multi-task-online","title":"Shopping MMLU: A Massive Multi-Task Online Shopping Benchmark for Large Language Models","date":"2024-10-28","arxiv_id":"2410.20745","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/shopping-mmlu-a-massive-multi-task-online#ran","syntology_url":"https://syntology.ai/paper/2410.20745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20745"}},"official":{"repos":["kl4805/shoppingmmlu"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/read-me-refactorizing-llms-as-router","slug":"read-me-refactorizing-llms-as-router","title":"Read-ME: Refactorizing LLMs as Router-Decoupled Mixture of Experts with System Co-Design","date":"2024-10-24","arxiv_id":"2410.19123","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":16,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/read-me-refactorizing-llms-as-router#ran","syntology_url":"https://syntology.ai/paper/2410.19123","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19123"}},"official":{"repos":["vita-group/read-me"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/bento-benchmark-task-reduction-with-in","slug":"bento-benchmark-task-reduction-with-in","title":"BenTo: Benchmark Task Reduction with In-Context Transferability","date":"2024-10-17","arxiv_id":"2410.13804","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bento-benchmark-task-reduction-with-in#ran","syntology_url":"https://syntology.ai/paper/2410.13804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13804"}},"official":{"repos":["tianyi-lab/bento"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/divide-reweight-and-conquer-a-logit","slug":"divide-reweight-and-conquer-a-logit","title":"Divide, Reweight, and Conquer: A Logit Arithmetic Approach for In-Context Learning","date":"2024-10-14","arxiv_id":"2410.10074","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/divide-reweight-and-conquer-a-logit#ran","syntology_url":"https://syntology.ai/paper/2410.10074","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10074"}},"official":{"repos":["chengsong-huang/lara"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lolcats-on-low-rank-linearizing-of-large","slug":"lolcats-on-low-rank-linearizing-of-large","title":"LoLCATs: On Low-Rank Linearizing of Large Language Models","date":"2024-10-14","arxiv_id":"2410.10254","repositories_listed":1,"syntology":{"n":32,"n_ran":23,"n_constructed":9,"n_ran_checked":13,"n_instrument":10,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"23 ran (of which 9 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 10 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/lolcats-on-low-rank-linearizing-of-large#ran","syntology_url":"https://syntology.ai/paper/2410.10254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10254"}},"official":{"repos":["hazyresearch/lolcats"],"state":"official (archive's flag): 23 ran","n_ran":23,"n_constructed":9,"n_ran_no_instrument_failure":13,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/comat-chain-of-mathematically-annotated","slug":"comat-chain-of-mathematically-annotated","title":"CoMAT: Chain of Mathematically Annotated Thought Improves Mathematical Reasoning","date":"2024-10-14","arxiv_id":"2410.10336","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/comat-chain-of-mathematically-annotated#ran","syntology_url":"https://syntology.ai/paper/2410.10336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10336"}},"official":{"repos":["joshuaongg21/comat"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mexa-multilingual-evaluation-of-english","slug":"mexa-multilingual-evaluation-of-english","title":"MEXA: Multilingual Evaluation of English-Centric LLMs via Cross-Lingual Alignment","date":"2024-10-08","arxiv_id":"2410.05873","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mexa-multilingual-evaluation-of-english#ran","syntology_url":"https://syntology.ai/paper/2410.05873","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05873"}},"official":{"repos":["cisnlp/mexa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/commonit-commonality-aware-instruction-tuning","slug":"commonit-commonality-aware-instruction-tuning","title":"CommonIT: Commonality-Aware Instruction Tuning for Large Language Models via Data Partitions","date":"2024-10-04","arxiv_id":"2410.03077","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/commonit-commonality-aware-instruction-tuning#ran","syntology_url":"https://syntology.ai/paper/2410.03077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03077"}},"official":{"repos":["raojay7/commonit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-topla-efficient-llm-ensemble-by","slug":"llm-topla-efficient-llm-ensemble-by","title":"LLM-TOPLA: Efficient LLM Ensemble by Maximising Diversity","date":"2024-10-04","arxiv_id":"2410.03953","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/llm-topla-efficient-llm-ensemble-by#ran","syntology_url":"https://syntology.ai/paper/2410.03953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03953"}},"official":{"repos":["git-disl/llm-topla"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/to-cot-or-not-to-cot-chain-of-thought-helps","slug":"to-cot-or-not-to-cot-chain-of-thought-helps","title":"To CoT or not to CoT? Chain-of-thought helps mainly on math and symbolic reasoning","date":"2024-09-18","arxiv_id":"2409.12183","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/to-cot-or-not-to-cot-chain-of-thought-helps#ran","syntology_url":"https://syntology.ai/paper/2409.12183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12183"}},"official":{"repos":["zayne-sprague/to-cot-or-not-to-cot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mmlu-pro-evaluating-higher-order-reasoning","slug":"mmlu-pro-evaluating-higher-order-reasoning","title":"MMLU-Pro+: Evaluating Higher-Order Reasoning and Shortcut Learning in LLMs","date":"2024-09-03","arxiv_id":"2409.02257","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mmlu-pro-evaluating-higher-order-reasoning#ran","syntology_url":"https://syntology.ai/paper/2409.02257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02257"}},"official":{"repos":["asgsaeid/mmlu-pro-plus"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-deeper-look-at-depth-pruning-of-llms","slug":"a-deeper-look-at-depth-pruning-of-llms","title":"A deeper look at depth pruning of LLMs","date":"2024-07-23","arxiv_id":"2407.16286","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":6,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-deeper-look-at-depth-pruning-of-llms#ran","syntology_url":"https://syntology.ai/paper/2407.16286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16286"}},"official":{"repos":["shoaibahmed/llm_depth_pruning"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/me-myself-and-ai-the-situational-awareness","slug":"me-myself-and-ai-the-situational-awareness","title":"Me, Myself, and AI: The Situational Awareness Dataset (SAD) for LLMs","date":"2024-07-05","arxiv_id":"2407.04694","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/me-myself-and-ai-the-situational-awareness#ran","syntology_url":"https://syntology.ai/paper/2407.04694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04694"}},"official":{"repos":["lrudl/sad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/texttt-metabench-a-sparse-benchmark-to","slug":"texttt-metabench-a-sparse-benchmark-to","title":"$\\texttt{metabench}$ -- A Sparse Benchmark to Measure General Ability in Large Language Models","date":"2024-07-04","arxiv_id":"2407.12844","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/texttt-metabench-a-sparse-benchmark-to#ran","syntology_url":"https://syntology.ai/paper/2407.12844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12844"}},"official":{"repos":["adkipnis/metabench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/empo-theory-driven-dataset-construction-for","slug":"empo-theory-driven-dataset-construction-for","title":"EmPO: Emotion Grounding for Empathetic Response Generation through Preference Optimization","date":"2024-06-27","arxiv_id":"2406.19071","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/empo-theory-driven-dataset-construction-for#ran","syntology_url":"https://syntology.ai/paper/2406.19071","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19071"}},"official":{"repos":["justtherightsize/empo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pruning-via-merging-compressing-llms-via","slug":"pruning-via-merging-compressing-llms-via","title":"Pruning via Merging: Compressing LLMs via Manifold Alignment Based Layer Merging","date":"2024-06-24","arxiv_id":"2406.16330","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":3,"n_instrument":7,"n_unverified":3,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 7 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pruning-via-merging-compressing-llms-via#ran","syntology_url":"https://syntology.ai/paper/2406.16330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16330"}},"official":{"repos":["sempraety/pruning-via-merging"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/training-free-exponential-extension-of","slug":"training-free-exponential-extension-of","title":"Training-Free Exponential Context Extension via Cascading KV Cache","date":"2024-06-24","arxiv_id":"2406.17808","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-free-exponential-extension-of#ran","syntology_url":"https://syntology.ai/paper/2406.17808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17808"}},"official":{"repos":["jeffwillette/cascading_kv_cache"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-matters-in-transformers-not-all","slug":"what-matters-in-transformers-not-all","title":"What Matters in Transformers? Not All Attention is Needed","date":"2024-06-22","arxiv_id":"2406.15786","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/what-matters-in-transformers-not-all#ran","syntology_url":"https://syntology.ai/paper/2406.15786","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15786"}},"official":{"repos":["case-lab-umd/llm-drop","shwai-he/llm-drop"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/livemind-low-latency-large-language-models","slug":"livemind-low-latency-large-language-models","title":"LiveMind: Low-latency Large Language Models with Simultaneous Inference","date":"2024-06-20","arxiv_id":"2406.14319","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/livemind-low-latency-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2406.14319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14319"}},"official":{"repos":["chuangtaochen-tum/livemind"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chatglm-a-family-of-large-language-models","slug":"chatglm-a-family-of-large-language-models","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","date":"2024-06-18","arxiv_id":"2406.12793","repositories_listed":7,"syntology":{"n":29,"n_ran":21,"n_constructed":0,"n_ran_checked":20,"n_instrument":1,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":20,"n_pointer_only":1,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/chatglm-a-family-of-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2406.12793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12793"}},"official":{"repos":["thudm/chatglm-6b"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/datacomp-lm-in-search-of-the-next-generation","slug":"datacomp-lm-in-search-of-the-next-generation","title":"DataComp-LM: In search of the next generation of training sets for language models","date":"2024-06-17","arxiv_id":"2406.11794","repositories_listed":3,"syntology":{"n":22,"n_ran":22,"n_constructed":0,"n_ran_checked":20,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":20,"n_pointer_only":1,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/datacomp-lm-in-search-of-the-next-generation#ran","syntology_url":"https://syntology.ai/paper/2406.11794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11794"}},"official":null}},{"url":"/paper/sharelora-parameter-efficient-and-robust","slug":"sharelora-parameter-efficient-and-robust","title":"ShareLoRA: Parameter Efficient and Robust Large Language Model Fine-tuning via Shared Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.10785","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sharelora-parameter-efficient-and-robust#ran","syntology_url":"https://syntology.ai/paper/2406.10785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10785"}},"official":{"repos":["Rain9876/ShareLoRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/are-we-done-with-mmlu","slug":"are-we-done-with-mmlu","title":"Are We Done with MMLU?","date":"2024-06-06","arxiv_id":"2406.04127","repositories_listed":3,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/are-we-done-with-mmlu#ran","syntology_url":"https://syntology.ai/paper/2406.04127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04127"}},"official":{"repos":["aryopg/mmlu-redux"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/do-large-language-models-perform-the-way","slug":"do-large-language-models-perform-the-way","title":"Do Large Language Models Perform the Way People Expect? Measuring the Human Generalization Function","date":"2024-06-03","arxiv_id":"2406.01382","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/do-large-language-models-perform-the-way#ran","syntology_url":"https://syntology.ai/paper/2406.01382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01382"}},"official":{"repos":["keyonvafa/human-generalization-llms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmlu-pro-a-more-robust-and-challenging-multi","slug":"mmlu-pro-a-more-robust-and-challenging-multi","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","date":"2024-06-03","arxiv_id":"2406.01574","repositories_listed":2,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":3,"n_instrument":9,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 9 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mmlu-pro-a-more-robust-and-challenging-multi#ran","syntology_url":"https://syntology.ai/paper/2406.01574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01574"}},"official":null}},{"url":"/paper/owlore-outlier-weighed-layerwise-sampled-low","slug":"owlore-outlier-weighed-layerwise-sampled-low","title":"OwLore: Outlier-weighed Layerwise Sampled Low-Rank Projection for Memory-Efficient LLM Fine-tuning","date":"2024-05-28","arxiv_id":"2405.18380","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/owlore-outlier-weighed-layerwise-sampled-low#ran","syntology_url":"https://syntology.ai/paper/2405.18380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18380"}},"official":{"repos":["pixeli99/owlore"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/efficient-multi-prompt-evaluation-of-llms","slug":"efficient-multi-prompt-evaluation-of-llms","title":"Efficient multi-prompt evaluation of LLMs","date":"2024-05-27","arxiv_id":"2405.17202","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":1,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-multi-prompt-evaluation-of-llms#ran","syntology_url":"https://syntology.ai/paper/2405.17202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17202"}},"official":{"repos":["felipemaiapolo/prompteval","microsoft/promptbench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-tuning-with-loss-over","slug":"instruction-tuning-with-loss-over","title":"Instruction Tuning With Loss Over Instructions","date":"2024-05-23","arxiv_id":"2405.14394","repositories_listed":1,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":20,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/instruction-tuning-with-loss-over#ran","syntology_url":"https://syntology.ai/paper/2405.14394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14394"}},"official":{"repos":["zhengxiangshi/instructionmodelling"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/layer-skip-enabling-early-exit-inference-and","slug":"layer-skip-enabling-early-exit-inference-and","title":"LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding","date":"2024-04-25","arxiv_id":"2404.16710","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/layer-skip-enabling-early-exit-inference-and#ran","syntology_url":"https://syntology.ai/paper/2404.16710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16710"}},"official":{"repos":["facebookresearch/layerskip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/make-your-llm-fully-utilize-the-context","slug":"make-your-llm-fully-utilize-the-context","title":"Make Your LLM Fully Utilize the Context","date":"2024-04-25","arxiv_id":"2404.16811","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/make-your-llm-fully-utilize-the-context#ran","syntology_url":"https://syntology.ai/paper/2404.16811","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16811"}},"official":{"repos":["microsoft/FILM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/advprompter-fast-adaptive-adversarial","slug":"advprompter-fast-adaptive-adversarial","title":"AdvPrompter: Fast Adaptive Adversarial Prompting for LLMs","date":"2024-04-21","arxiv_id":"2404.16873","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/advprompter-fast-adaptive-adversarial#ran","syntology_url":"https://syntology.ai/paper/2404.16873","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16873"}},"official":{"repos":["facebookresearch/advprompter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-training-small-base-lms-with-fewer-tokens","slug":"pre-training-small-base-lms-with-fewer-tokens","title":"Inheritune: Training Smaller Yet More Attentive Language Models","date":"2024-04-12","arxiv_id":"2404.08634","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pre-training-small-base-lms-with-fewer-tokens#ran","syntology_url":"https://syntology.ai/paper/2404.08634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08634"}},"official":{"repos":["sanyalsunny111/llm-inheritune"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/post-hoc-reversal-are-we-selecting-models","slug":"post-hoc-reversal-are-we-selecting-models","title":"Post-Hoc Reversal: Are We Selecting Models Prematurely?","date":"2024-04-11","arxiv_id":"2404.07815","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/post-hoc-reversal-are-we-selecting-models#ran","syntology_url":"https://syntology.ai/paper/2404.07815","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07815"}},"official":{"repos":["rishabh-ranjan/post-hoc-reversal"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/biomedlm-a-2-7b-parameter-language-model","slug":"biomedlm-a-2-7b-parameter-language-model","title":"BioMedLM: A 2.7B Parameter Language Model Trained On Biomedical Text","date":"2024-03-27","arxiv_id":"2403.18421","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/biomedlm-a-2-7b-parameter-language-model#ran","syntology_url":"https://syntology.ai/paper/2403.18421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18421"}},"official":{"repos":["stanford-crfm/biomedlm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sotopia-p-interactive-learning-of-socially","slug":"sotopia-p-interactive-learning-of-socially","title":"SOTOPIA-$π$: Interactive Learning of Socially Intelligent Language Agents","date":"2024-03-13","arxiv_id":"2403.08715","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sotopia-p-interactive-learning-of-socially#ran","syntology_url":"https://syntology.ai/paper/2403.08715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.08715"}},"official":{"repos":["sotopia-lab/sotopia-pi"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unfamiliar-finetuning-examples-control-how","slug":"unfamiliar-finetuning-examples-control-how","title":"Unfamiliar Finetuning Examples Control How Language Models Hallucinate","date":"2024-03-08","arxiv_id":"2403.05612","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unfamiliar-finetuning-examples-control-how#ran","syntology_url":"https://syntology.ai/paper/2403.05612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05612"}},"official":{"repos":["katiekang1998/llm_hallucinations"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/yi-open-foundation-models-by-01-ai","slug":"yi-open-foundation-models-by-01-ai","title":"Yi: Open Foundation Models by 01.AI","date":"2024-03-07","arxiv_id":"2403.04652","repositories_listed":1,"syntology":{"n":8,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/yi-open-foundation-models-by-01-ai#ran","syntology_url":"https://syntology.ai/paper/2403.04652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04652"}},"official":{"repos":["01-ai/yi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/chatmusician-understanding-and-generating","slug":"chatmusician-understanding-and-generating","title":"ChatMusician: Understanding and Generating Music Intrinsically with LLM","date":"2024-02-25","arxiv_id":"2402.16153","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/chatmusician-understanding-and-generating#ran","syntology_url":"https://syntology.ai/paper/2402.16153","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16153"}},"official":{"repos":["hf-lin/ChatMusician"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tinybenchmarks-evaluating-llms-with-fewer","slug":"tinybenchmarks-evaluating-llms-with-fewer","title":"tinyBenchmarks: evaluating LLMs with fewer examples","date":"2024-02-22","arxiv_id":"2402.14992","repositories_listed":4,"syntology":{"n":39,"n_ran":25,"n_constructed":0,"n_ran_checked":25,"n_instrument":0,"n_unverified":14,"n_honours":0,"n_violates":1,"n_no_contract":24,"n_pointer_only":17,"phrase":"25 ran (of which 0 constructed an object rather than computing a result; 25 with no instrument failure: 0 honoured, 1 violated, 24 with no contract checked; 0 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/tinybenchmarks-evaluating-llms-with-fewer#ran","syntology_url":"https://syntology.ai/paper/2402.14992","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14992"}},"official":{"repos":["felipemaiapolo/efficbench","felipemaiapolo/tinybenchmarks"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":12,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/accurate-lora-finetuning-quantization-of-llms","slug":"accurate-lora-finetuning-quantization-of-llms","title":"Accurate LoRA-Finetuning Quantization of LLMs via Information Retention","date":"2024-02-08","arxiv_id":"2402.05445","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accurate-lora-finetuning-quantization-of-llms#ran","syntology_url":"https://syntology.ai/paper/2402.05445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05445"}},"official":{"repos":["htqin/ir-qlora"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-benchmarks-are-targets-revealing-the","slug":"when-benchmarks-are-targets-revealing-the","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","date":"2024-02-01","arxiv_id":"2402.01781","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/when-benchmarks-are-targets-revealing-the#ran","syntology_url":"https://syntology.ai/paper/2402.01781","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01781"}},"official":{"repos":["national-center-for-ai-saudi-arabia/lm-evaluation-harness"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leeroo-orchestrator-elevating-llms","slug":"leeroo-orchestrator-elevating-llms","title":"Routoo: Learning to Route to Large Language Models Effectively","date":"2024-01-25","arxiv_id":"2401.13979","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leeroo-orchestrator-elevating-llms#ran","syntology_url":"https://syntology.ai/paper/2401.13979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13979"}},"official":{"repos":["leeroo-ai/leeroo_orchestrator"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/eq-bench-an-emotional-intelligence-benchmark","slug":"eq-bench-an-emotional-intelligence-benchmark","title":"EQ-Bench: An Emotional Intelligence Benchmark for Large Language Models","date":"2023-12-11","arxiv_id":"2312.06281","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eq-bench-an-emotional-intelligence-benchmark#ran","syntology_url":"https://syntology.ai/paper/2312.06281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06281"}},"official":{"repos":["eq-bench/eq-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-online-data-mixing-for-language","slug":"efficient-online-data-mixing-for-language","title":"Efficient Online Data Mixing For Language Model Pre-Training","date":"2023-12-05","arxiv_id":"2312.02406","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-online-data-mixing-for-language#ran","syntology_url":"https://syntology.ai/paper/2312.02406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02406"}},"official":null}},{"url":"/paper/prompt-optimization-via-adversarial-in","slug":"prompt-optimization-via-adversarial-in","title":"Prompt Optimization via Adversarial In-Context Learning","date":"2023-12-05","arxiv_id":"2312.02614","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prompt-optimization-via-adversarial-in#ran","syntology_url":"https://syntology.ai/paper/2312.02614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02614"}},"official":{"repos":["zhaoyiran924/adv-in-context-learning"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/compeft-compression-for-communicating","slug":"compeft-compression-for-communicating","title":"ComPEFT: Compression for Communicating Parameter Efficient Updates via Sparsification and Quantization","date":"2023-11-22","arxiv_id":"2311.13171","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/compeft-compression-for-communicating#ran","syntology_url":"https://syntology.ai/paper/2311.13171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13171"}},"official":{"repos":["prateeky2806/compeft"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lm-cocktail-resilient-tuning-of-language","slug":"lm-cocktail-resilient-tuning-of-language","title":"LM-Cocktail: Resilient Tuning of Language Models via Model Merging","date":"2023-11-22","arxiv_id":"2311.13534","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lm-cocktail-resilient-tuning-of-language#ran","syntology_url":"https://syntology.ai/paper/2311.13534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13534"}},"official":{"repos":["flagopen/flagembedding"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/medagents-large-language-models-as","slug":"medagents-large-language-models-as","title":"MedAgents: Large Language Models as Collaborators for Zero-shot Medical Reasoning","date":"2023-11-16","arxiv_id":"2311.10537","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/medagents-large-language-models-as#ran","syntology_url":"https://syntology.ai/paper/2311.10537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10537"}},"official":{"repos":["gersteinlab/medagents"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-benchmark-and-contamination-for","slug":"rethinking-benchmark-and-contamination-for","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","date":"2023-11-08","arxiv_id":"2311.04850","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rethinking-benchmark-and-contamination-for#ran","syntology_url":"https://syntology.ai/paper/2311.04850","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04850"}},"official":{"repos":["lm-sys/llm-decontaminator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/compresso-structured-pruning-with","slug":"compresso-structured-pruning-with","title":"Compresso: Structured Pruning with Collaborative Prompting Learns Compact Large Language Models","date":"2023-10-08","arxiv_id":"2310.05015","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/compresso-structured-pruning-with#ran","syntology_url":"https://syntology.ai/paper/2310.05015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05015"}},"official":{"repos":["microsoft/moonlit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-llm-agent-network-an-llm-agent","slug":"dynamic-llm-agent-network-an-llm-agent","title":"A Dynamic LLM-Powered Agent Network for Task-Oriented Agent Collaboration","date":"2023-10-03","arxiv_id":"2310.02170","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-llm-agent-network-an-llm-agent#ran","syntology_url":"https://syntology.ai/paper/2310.02170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02170"}},"official":{"repos":["salt-nlp/dylan"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-channel-dimensions-to-isolate","slug":"rethinking-channel-dimensions-to-isolate","title":"Rethinking Channel Dimensions to Isolate Outliers for Low-bit Weight Quantization of Large Language Models","date":"2023-09-27","arxiv_id":"2309.15531","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rethinking-channel-dimensions-to-isolate#ran","syntology_url":"https://syntology.ai/paper/2309.15531","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15531"}},"official":{"repos":["johnheo/adadim-llm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/augmentation-adapted-retriever-improves","slug":"augmentation-adapted-retriever-improves","title":"Augmentation-Adapted Retriever Improves Generalization of Language Models as Generic Plug-In","date":"2023-05-27","arxiv_id":"2305.17331","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":1,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":2,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/augmentation-adapted-retriever-improves#ran","syntology_url":"https://syntology.ai/paper/2305.17331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17331"}},"official":{"repos":["openmatch/augmentation-adapted-retriever"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/the-art-of-socratic-questioning-zero-shot","slug":"the-art-of-socratic-questioning-zero-shot","title":"The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models","date":"2023-05-24","arxiv_id":"2305.14999","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/the-art-of-socratic-questioning-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2305.14999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14999"}},"official":{"repos":["vt-nlp/socratic-questioning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/from-zero-to-hero-examining-the-power-of","slug":"from-zero-to-hero-examining-the-power-of","title":"From Zero to Hero: Examining the Power of Symbolic Tasks in Instruction Tuning","date":"2023-04-17","arxiv_id":"2304.07995","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/from-zero-to-hero-examining-the-power-of#ran","syntology_url":"https://syntology.ai/paper/2304.07995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.07995"}},"official":{"repos":["sail-sg/symbolic-instruction-tuning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/art-automatic-multi-step-reasoning-and-tool","slug":"art-automatic-multi-step-reasoning-and-tool","title":"ART: Automatic multi-step reasoning and tool-use for large language models","date":"2023-03-16","arxiv_id":"2303.09014","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/art-automatic-multi-step-reasoning-and-tool#ran","syntology_url":"https://syntology.ai/paper/2303.09014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.09014"}},"official":null}},{"url":"/paper/replug-retrieval-augmented-black-box-language","slug":"replug-retrieval-augmented-black-box-language","title":"REPLUG: Retrieval-Augmented Black-Box Language Models","date":"2023-01-30","arxiv_id":"2301.12652","repositories_listed":3,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/replug-retrieval-augmented-black-box-language#ran","syntology_url":"https://syntology.ai/paper/2301.12652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12652"}},"official":null}},{"url":"/paper/galactica-a-large-language-model-for-science-1","slug":"galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","arxiv_id":"2211.09085","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/galactica-a-large-language-model-for-science-1#ran","syntology_url":"https://syntology.ai/paper/2211.09085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09085"}},"official":{"repos":["paperswithcode/galai"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-instruction-finetuned-language-models","slug":"scaling-instruction-finetuned-language-models","title":"Scaling Instruction-Finetuned Language Models","date":"2022-10-20","arxiv_id":"2210.11416","repositories_listed":9,"syntology":{"n":17,"n_ran":8,"n_constructed":1,"n_ran_checked":1,"n_instrument":7,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 7 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/scaling-instruction-finetuned-language-models#ran","syntology_url":"https://syntology.ai/paper/2210.11416","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11416"}},"official":null}},{"url":"/paper/few-shot-learning-with-retrieval-augmented","slug":"few-shot-learning-with-retrieval-augmented","title":"Atlas: Few-shot Learning with Retrieval Augmented Language Models","date":"2022-08-05","arxiv_id":"2208.03299","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/few-shot-learning-with-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2208.03299","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.03299"}},"official":{"repos":["facebookresearch/atlas"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-language-learning-paradigms","slug":"unifying-language-learning-paradigms","title":"UL2: Unifying Language Learning Paradigms","date":"2022-05-10","arxiv_id":"2205.05131","repositories_listed":2,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unifying-language-learning-paradigms#ran","syntology_url":"https://syntology.ai/paper/2205.05131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05131"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/training-compute-optimal-large-language","slug":"training-compute-optimal-large-language","title":"Training Compute-Optimal Large Language Models","date":"2022-03-29","arxiv_id":"2203.15556","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":3,"n_ran_checked":5,"n_instrument":3,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/training-compute-optimal-large-language#ran","syntology_url":"https://syntology.ai/paper/2203.15556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.15556"}},"official":null}}],"record_sha256":"bc8878c58e827b75942ba960e951c08184c7fa0d2fedf33ff47176d72586dce3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}