{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mmlu/papers/2","list_of":"/task/mmlu","task":"MMLU","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":340,"counts":{"archive_papers_tagged":340,"with_a_code_link":156,"where_syntology_ran_a_sample":78,"not_listed_spam_title":0,"listed":340,"listed_where_code_ran":78,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":60,"every_run_a_failure_of_syntologys_instrument":18,"listed_with_a_run_with_no_instrument_failure":60,"listed_every_run_a_failure_of_syntologys_instrument":18,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mmlu","prev":"/task/mmlu","next":"/task/mmlu/papers/3","papers":[{"url":"/paper/me-myself-and-ai-the-situational-awareness","slug":"me-myself-and-ai-the-situational-awareness","title":"Me, Myself, and AI: The Situational Awareness Dataset (SAD) for LLMs","date":"2024-07-05","arxiv_id":"2407.04694","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/me-myself-and-ai-the-situational-awareness#ran","syntology_url":"https://syntology.ai/paper/2407.04694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04694"}},"official":{"repos":["lrudl/sad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/texttt-metabench-a-sparse-benchmark-to","slug":"texttt-metabench-a-sparse-benchmark-to","title":"$\\texttt{metabench}$ -- A Sparse Benchmark to Measure General Ability in Large Language Models","date":"2024-07-04","arxiv_id":"2407.12844","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/texttt-metabench-a-sparse-benchmark-to#ran","syntology_url":"https://syntology.ai/paper/2407.12844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12844"}},"official":{"repos":["adkipnis/metabench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/empo-theory-driven-dataset-construction-for","slug":"empo-theory-driven-dataset-construction-for","title":"EmPO: Emotion Grounding for Empathetic Response Generation through Preference Optimization","date":"2024-06-27","arxiv_id":"2406.19071","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/empo-theory-driven-dataset-construction-for#ran","syntology_url":"https://syntology.ai/paper/2406.19071","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19071"}},"official":{"repos":["justtherightsize/empo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-fineweb-datasets-decanting-the-web-for","slug":"the-fineweb-datasets-decanting-the-web-for","title":"The FineWeb Datasets: Decanting the Web for the Finest Text Data at Scale","date":"2024-06-25","arxiv_id":"2406.17557","repositories_listed":1,"syntology":null},{"url":"/paper/pruning-via-merging-compressing-llms-via","slug":"pruning-via-merging-compressing-llms-via","title":"Pruning via Merging: Compressing LLMs via Manifold Alignment Based Layer Merging","date":"2024-06-24","arxiv_id":"2406.16330","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":3,"n_instrument":7,"n_unverified":3,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 7 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pruning-via-merging-compressing-llms-via#ran","syntology_url":"https://syntology.ai/paper/2406.16330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16330"}},"official":{"repos":["sempraety/pruning-via-merging"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/training-free-exponential-extension-of","slug":"training-free-exponential-extension-of","title":"Training-Free Exponential Context Extension via Cascading KV Cache","date":"2024-06-24","arxiv_id":"2406.17808","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-free-exponential-extension-of#ran","syntology_url":"https://syntology.ai/paper/2406.17808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17808"}},"official":{"repos":["jeffwillette/cascading_kv_cache"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crosslingual-capabilities-and-knowledge","slug":"crosslingual-capabilities-and-knowledge","title":"Crosslingual Capabilities and Knowledge Barriers in Multilingual Large Language Models","date":"2024-06-23","arxiv_id":"2406.16135","repositories_listed":1,"syntology":null},{"url":"/paper/inference-time-decontamination-reusing-leaked","slug":"inference-time-decontamination-reusing-leaked","title":"Inference-Time Decontamination: Reusing Leaked Benchmarks for Large Language Model Evaluation","date":"2024-06-20","arxiv_id":"2406.13990","repositories_listed":1,"syntology":null},{"url":"/paper/livemind-low-latency-large-language-models","slug":"livemind-low-latency-large-language-models","title":"LiveMind: Low-latency Large Language Models with Simultaneous Inference","date":"2024-06-20","arxiv_id":"2406.14319","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/livemind-low-latency-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2406.14319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14319"}},"official":{"repos":["chuangtaochen-tum/livemind"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/input-conditioned-graph-generation-for","slug":"input-conditioned-graph-generation-for","title":"Input Conditioned Graph Generation for Language Agents","date":"2024-06-17","arxiv_id":"2406.11555","repositories_listed":1,"syntology":null},{"url":"/paper/sharelora-parameter-efficient-and-robust","slug":"sharelora-parameter-efficient-and-robust","title":"ShareLoRA: Parameter Efficient and Robust Large Language Model Fine-tuning via Shared Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.10785","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sharelora-parameter-efficient-and-robust#ran","syntology_url":"https://syntology.ai/paper/2406.10785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10785"}},"official":{"repos":["Rain9876/ShareLoRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-of-mamba-based-language","slug":"an-empirical-study-of-mamba-based-language","title":"An Empirical Study of Mamba-based Language Models","date":"2024-06-12","arxiv_id":"2406.07887","repositories_listed":1,"syntology":null},{"url":"/paper/do-large-language-models-perform-the-way","slug":"do-large-language-models-perform-the-way","title":"Do Large Language Models Perform the Way People Expect? Measuring the Human Generalization Function","date":"2024-06-03","arxiv_id":"2406.01382","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/do-large-language-models-perform-the-way#ran","syntology_url":"https://syntology.ai/paper/2406.01382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01382"}},"official":{"repos":["keyonvafa/human-generalization-llms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-tuning-with-loss-over","slug":"instruction-tuning-with-loss-over","title":"Instruction Tuning With Loss Over Instructions","date":"2024-05-23","arxiv_id":"2405.14394","repositories_listed":1,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":20,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/instruction-tuning-with-loss-over#ran","syntology_url":"https://syntology.ai/paper/2405.14394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14394"}},"official":{"repos":["zhengxiangshi/instructionmodelling"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-moe-and-dense-speed-accuracy","slug":"revisiting-moe-and-dense-speed-accuracy","title":"Revisiting MoE and Dense Speed-Accuracy Comparisons for LLM Training","date":"2024-05-23","arxiv_id":"2405.15052","repositories_listed":1,"syntology":null},{"url":"/paper/layer-skip-enabling-early-exit-inference-and","slug":"layer-skip-enabling-early-exit-inference-and","title":"LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding","date":"2024-04-25","arxiv_id":"2404.16710","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/layer-skip-enabling-early-exit-inference-and#ran","syntology_url":"https://syntology.ai/paper/2404.16710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16710"}},"official":{"repos":["facebookresearch/layerskip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/advprompter-fast-adaptive-adversarial","slug":"advprompter-fast-adaptive-adversarial","title":"AdvPrompter: Fast Adaptive Adversarial Prompting for LLMs","date":"2024-04-21","arxiv_id":"2404.16873","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/advprompter-fast-adaptive-adversarial#ran","syntology_url":"https://syntology.ai/paper/2404.16873","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16873"}},"official":{"repos":["facebookresearch/advprompter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-training-small-base-lms-with-fewer-tokens","slug":"pre-training-small-base-lms-with-fewer-tokens","title":"Inheritune: Training Smaller Yet More Attentive Language Models","date":"2024-04-12","arxiv_id":"2404.08634","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pre-training-small-base-lms-with-fewer-tokens#ran","syntology_url":"https://syntology.ai/paper/2404.08634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08634"}},"official":{"repos":["sanyalsunny111/llm-inheritune"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/post-hoc-reversal-are-we-selecting-models","slug":"post-hoc-reversal-are-we-selecting-models","title":"Post-Hoc Reversal: Are We Selecting Models Prematurely?","date":"2024-04-11","arxiv_id":"2404.07815","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/post-hoc-reversal-are-we-selecting-models#ran","syntology_url":"https://syntology.ai/paper/2404.07815","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07815"}},"official":{"repos":["rishabh-ranjan/post-hoc-reversal"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flawn-t5-an-empirical-examination-of","slug":"flawn-t5-an-empirical-examination-of","title":"LawInstruct: A Resource for Studying Language Model Adaptation to the Legal Domain","date":"2024-04-02","arxiv_id":"2404.02127","repositories_listed":1,"syntology":null},{"url":"/paper/biomedlm-a-2-7b-parameter-language-model","slug":"biomedlm-a-2-7b-parameter-language-model","title":"BioMedLM: A 2.7B Parameter Language Model Trained On Biomedical Text","date":"2024-03-27","arxiv_id":"2403.18421","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/biomedlm-a-2-7b-parameter-language-model#ran","syntology_url":"https://syntology.ai/paper/2403.18421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18421"}},"official":{"repos":["stanford-crfm/biomedlm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lisa-layerwise-importance-sampling-for-memory","slug":"lisa-layerwise-importance-sampling-for-memory","title":"LISA: Layerwise Importance Sampling for Memory-Efficient Large Language Model Fine-Tuning","date":"2024-03-26","arxiv_id":"2403.17919","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/lisa-layerwise-importance-sampling-for-memory#ran","syntology_url":"https://syntology.ai/paper/2403.17919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17919"}},"official":{"repos":["optimalscale/lmflow"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/sotopia-p-interactive-learning-of-socially","slug":"sotopia-p-interactive-learning-of-socially","title":"SOTOPIA-$π$: Interactive Learning of Socially Intelligent Language Agents","date":"2024-03-13","arxiv_id":"2403.08715","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sotopia-p-interactive-learning-of-socially#ran","syntology_url":"https://syntology.ai/paper/2403.08715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.08715"}},"official":{"repos":["sotopia-lab/sotopia-pi"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unfamiliar-finetuning-examples-control-how","slug":"unfamiliar-finetuning-examples-control-how","title":"Unfamiliar Finetuning Examples Control How Language Models Hallucinate","date":"2024-03-08","arxiv_id":"2403.05612","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unfamiliar-finetuning-examples-control-how#ran","syntology_url":"https://syntology.ai/paper/2403.05612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05612"}},"official":{"repos":["katiekang1998/llm_hallucinations"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/yi-open-foundation-models-by-01-ai","slug":"yi-open-foundation-models-by-01-ai","title":"Yi: Open Foundation Models by 01.AI","date":"2024-03-07","arxiv_id":"2403.04652","repositories_listed":1,"syntology":{"n":8,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/yi-open-foundation-models-by-01-ai#ran","syntology_url":"https://syntology.ai/paper/2403.04652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04652"}},"official":{"repos":["01-ai/yi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/to-generate-or-to-retrieve-on-the","slug":"to-generate-or-to-retrieve-on-the","title":"To Generate or to Retrieve? On the Effectiveness of Artificial Contexts for Medical Open-Domain Question Answering","date":"2024-03-04","arxiv_id":"2403.01924","repositories_listed":1,"syntology":null},{"url":"/paper/mathsensei-a-tool-augmented-large-language","slug":"mathsensei-a-tool-augmented-large-language","title":"MATHSENSEI: A Tool-Augmented Large Language Model for Mathematical Reasoning","date":"2024-02-27","arxiv_id":"2402.17231","repositories_listed":1,"syntology":null},{"url":"/paper/unleashing-the-potential-of-large-language-1","slug":"unleashing-the-potential-of-large-language-1","title":"Unleashing the Potential of Large Language Models as Prompt Optimizers: Analogical Analysis with Gradient-based Model Optimizers","date":"2024-02-27","arxiv_id":"2402.17564","repositories_listed":1,"syntology":null},{"url":"/paper/chatmusician-understanding-and-generating","slug":"chatmusician-understanding-and-generating","title":"ChatMusician: Understanding and Generating Music Intrinsically with LLM","date":"2024-02-25","arxiv_id":"2402.16153","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/chatmusician-understanding-and-generating#ran","syntology_url":"https://syntology.ai/paper/2402.16153","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16153"}},"official":{"repos":["hf-lin/ChatMusician"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/accurate-lora-finetuning-quantization-of-llms","slug":"accurate-lora-finetuning-quantization-of-llms","title":"Accurate LoRA-Finetuning Quantization of LLMs via Information Retention","date":"2024-02-08","arxiv_id":"2402.05445","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accurate-lora-finetuning-quantization-of-llms#ran","syntology_url":"https://syntology.ai/paper/2402.05445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05445"}},"official":{"repos":["htqin/ir-qlora"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-benchmarks-are-targets-revealing-the","slug":"when-benchmarks-are-targets-revealing-the","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","date":"2024-02-01","arxiv_id":"2402.01781","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/when-benchmarks-are-targets-revealing-the#ran","syntology_url":"https://syntology.ai/paper/2402.01781","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01781"}},"official":{"repos":["national-center-for-ai-saudi-arabia/lm-evaluation-harness"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leeroo-orchestrator-elevating-llms","slug":"leeroo-orchestrator-elevating-llms","title":"Routoo: Learning to Route to Large Language Models Effectively","date":"2024-01-25","arxiv_id":"2401.13979","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leeroo-orchestrator-elevating-llms#ran","syntology_url":"https://syntology.ai/paper/2401.13979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13979"}},"official":{"repos":["leeroo-ai/leeroo_orchestrator"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/aurora-activating-chinese-chat-capability-for","slug":"aurora-activating-chinese-chat-capability-for","title":"Aurora:Activating Chinese chat capability for Mixtral-8x7B sparse Mixture-of-Experts through Instruction-Tuning","date":"2023-12-22","arxiv_id":"2312.14557","repositories_listed":1,"syntology":null},{"url":"/paper/gemini-a-family-of-highly-capable-multimodal-1","slug":"gemini-a-family-of-highly-capable-multimodal-1","title":"Gemini: A Family of Highly Capable Multimodal Models","date":"2023-12-19","arxiv_id":"2312.11805","repositories_listed":1,"syntology":null},{"url":"/paper/eq-bench-an-emotional-intelligence-benchmark","slug":"eq-bench-an-emotional-intelligence-benchmark","title":"EQ-Bench: An Emotional Intelligence Benchmark for Large Language Models","date":"2023-12-11","arxiv_id":"2312.06281","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eq-bench-an-emotional-intelligence-benchmark#ran","syntology_url":"https://syntology.ai/paper/2312.06281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06281"}},"official":{"repos":["eq-bench/eq-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-online-data-mixing-for-language","slug":"efficient-online-data-mixing-for-language","title":"Efficient Online Data Mixing For Language Model Pre-Training","date":"2023-12-05","arxiv_id":"2312.02406","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-online-data-mixing-for-language#ran","syntology_url":"https://syntology.ai/paper/2312.02406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02406"}},"official":null}},{"url":"/paper/prompt-optimization-via-adversarial-in","slug":"prompt-optimization-via-adversarial-in","title":"Prompt Optimization via Adversarial In-Context Learning","date":"2023-12-05","arxiv_id":"2312.02614","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prompt-optimization-via-adversarial-in#ran","syntology_url":"https://syntology.ai/paper/2312.02614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02614"}},"official":{"repos":["zhaoyiran924/adv-in-context-learning"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/arcmmlu-a-library-and-information-science","slug":"arcmmlu-a-library-and-information-science","title":"ArcMMLU: A Library and Information Science Benchmark for Large Language Models","date":"2023-11-30","arxiv_id":"2311.18658","repositories_listed":1,"syntology":null},{"url":"/paper/compeft-compression-for-communicating","slug":"compeft-compression-for-communicating","title":"ComPEFT: Compression for Communicating Parameter Efficient Updates via Sparsification and Quantization","date":"2023-11-22","arxiv_id":"2311.13171","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/compeft-compression-for-communicating#ran","syntology_url":"https://syntology.ai/paper/2311.13171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13171"}},"official":{"repos":["prateeky2806/compeft"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lm-cocktail-resilient-tuning-of-language","slug":"lm-cocktail-resilient-tuning-of-language","title":"LM-Cocktail: Resilient Tuning of Language Models via Model Merging","date":"2023-11-22","arxiv_id":"2311.13534","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lm-cocktail-resilient-tuning-of-language#ran","syntology_url":"https://syntology.ai/paper/2311.13534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13534"}},"official":{"repos":["flagopen/flagembedding"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/medagents-large-language-models-as","slug":"medagents-large-language-models-as","title":"MedAgents: Large Language Models as Collaborators for Zero-shot Medical Reasoning","date":"2023-11-16","arxiv_id":"2311.10537","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/medagents-large-language-models-as#ran","syntology_url":"https://syntology.ai/paper/2311.10537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10537"}},"official":{"repos":["gersteinlab/medagents"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-benchmark-and-contamination-for","slug":"rethinking-benchmark-and-contamination-for","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","date":"2023-11-08","arxiv_id":"2311.04850","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rethinking-benchmark-and-contamination-for#ran","syntology_url":"https://syntology.ai/paper/2311.04850","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04850"}},"official":{"repos":["lm-sys/llm-decontaminator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-open-source-data-contamination-report-for","slug":"an-open-source-data-contamination-report-for","title":"An Open Source Data Contamination Report for Large Language Models","date":"2023-10-26","arxiv_id":"2310.17589","repositories_listed":1,"syntology":null},{"url":"/paper/instruction-tuning-with-human-curriculum","slug":"instruction-tuning-with-human-curriculum","title":"Instruction Tuning with Human Curriculum","date":"2023-10-14","arxiv_id":"2310.09518","repositories_listed":1,"syntology":null},{"url":"/paper/compresso-structured-pruning-with","slug":"compresso-structured-pruning-with","title":"Compresso: Structured Pruning with Collaborative Prompting Learns Compact Large Language Models","date":"2023-10-08","arxiv_id":"2310.05015","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/compresso-structured-pruning-with#ran","syntology_url":"https://syntology.ai/paper/2310.05015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05015"}},"official":{"repos":["microsoft/moonlit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-llm-agent-network-an-llm-agent","slug":"dynamic-llm-agent-network-an-llm-agent","title":"A Dynamic LLM-Powered Agent Network for Task-Oriented Agent Collaboration","date":"2023-10-03","arxiv_id":"2310.02170","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-llm-agent-network-an-llm-agent#ran","syntology_url":"https://syntology.ai/paper/2310.02170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02170"}},"official":{"repos":["salt-nlp/dylan"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-channel-dimensions-to-isolate","slug":"rethinking-channel-dimensions-to-isolate","title":"Rethinking Channel Dimensions to Isolate Outliers for Low-bit Weight Quantization of Large Language Models","date":"2023-09-27","arxiv_id":"2309.15531","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rethinking-channel-dimensions-to-isolate#ran","syntology_url":"https://syntology.ai/paper/2309.15531","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15531"}},"official":{"repos":["johnheo/adadim-llm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/openba-an-open-sourced-15b-bilingual","slug":"openba-an-open-sourced-15b-bilingual","title":"OpenBA: An Open-sourced 15B Bilingual Asymmetric seq2seq Model Pre-trained from Scratch","date":"2023-09-19","arxiv_id":"2309.10706","repositories_listed":1,"syntology":null},{"url":"/paper/empowering-cross-lingual-abilities-of","slug":"empowering-cross-lingual-abilities-of","title":"Empowering Cross-lingual Abilities of Instruction-tuned Large Language Models by Translation-following demonstrations","date":"2023-08-27","arxiv_id":"2308.14186","repositories_listed":1,"syntology":null},{"url":"/paper/the-art-of-socratic-questioning-zero-shot","slug":"the-art-of-socratic-questioning-zero-shot","title":"The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models","date":"2023-05-24","arxiv_id":"2305.14999","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/the-art-of-socratic-questioning-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2305.14999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14999"}},"official":{"repos":["vt-nlp/socratic-questioning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-generated-pretraining-signals-improves","slug":"model-generated-pretraining-signals-improves","title":"Model-Generated Pretraining Signals Improves Zero-Shot Generalization of Text-to-Text Transformers","date":"2023-05-21","arxiv_id":"2305.12567","repositories_listed":1,"syntology":null},{"url":"/paper/towards-expert-level-medical-question","slug":"towards-expert-level-medical-question","title":"Towards Expert-Level Medical Question Answering with Large Language Models","date":"2023-05-16","arxiv_id":"2305.09617","repositories_listed":1,"syntology":null},{"url":"/paper/from-zero-to-hero-examining-the-power-of","slug":"from-zero-to-hero-examining-the-power-of","title":"From Zero to Hero: Examining the Power of Symbolic Tasks in Instruction Tuning","date":"2023-04-17","arxiv_id":"2304.07995","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/from-zero-to-hero-examining-the-power-of#ran","syntology_url":"https://syntology.ai/paper/2304.07995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.07995"}},"official":{"repos":["sail-sg/symbolic-instruction-tuning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-inconsistencies-of-conditionals","slug":"on-the-inconsistencies-of-conditionals","title":"Inconsistencies in Masked Language Models","date":"2022-12-30","arxiv_id":"2301.00068","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-encode-clinical","slug":"large-language-models-encode-clinical","title":"Large Language Models Encode Clinical Knowledge","date":"2022-12-26","arxiv_id":"2212.13138","repositories_listed":1,"syntology":null},{"url":"/paper/galactica-a-large-language-model-for-science-1","slug":"galactica-a-large-language-model-for-science-1","title":"Galactica: A Large Language Model for Science","date":"2022-11-16","arxiv_id":"2211.09085","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/galactica-a-large-language-model-for-science-1#ran","syntology_url":"https://syntology.ai/paper/2211.09085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09085"}},"official":{"repos":["paperswithcode/galai"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":null,"slug":"learning-what-matters-probabilistic-task","title":"Learning What Matters: Probabilistic Task Selection via Mutual Information for Model Finetuning","date":"2025-07-16","arxiv_id":"2507.12612","repositories_listed":0,"syntology":null},{"url":null,"slug":"lizard-an-efficient-linearization-framework","title":"Lizard: An Efficient Linearization Framework for Large Language Models","date":"2025-07-11","arxiv_id":"2507.09025","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-external-tools-with-large","title":"Integrating External Tools with Large Language Models to Improve Accuracy","date":"2025-07-09","arxiv_id":"2507.08034","repositories_listed":0,"syntology":null},{"url":"/paper/reinforcement-fine-tuning-naturally-mitigates","slug":"reinforcement-fine-tuning-naturally-mitigates","title":"Reinforcement Fine-Tuning Naturally Mitigates Forgetting in Continual Post-Training","date":"2025-07-07","arxiv_id":"2507.05386","repositories_listed":0,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reinforcement-fine-tuning-naturally-mitigates#ran","syntology_url":"https://syntology.ai/paper/2507.05386","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.05386"}},"official":null}},{"url":null,"slug":"biomed-enriched-a-biomedical-dataset-enriched","title":"Biomed-Enriched: A Biomedical Dataset Enriched with LLMs for Pretraining and Extracting Rare and Hidden Content","date":"2025-06-25","arxiv_id":"2506.20331","repositories_listed":0,"syntology":null},{"url":null,"slug":"enterprise-large-language-model-evaluation","title":"Enterprise Large Language Model Evaluation Benchmark","date":"2025-06-25","arxiv_id":"2506.20274","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-lingual-functional-evaluation-for-large","title":"Multi-lingual Functional Evaluation for Large Language Models","date":"2025-06-25","arxiv_id":"2506.20793","repositories_listed":0,"syntology":null},{"url":null,"slug":"gazal-r1-achieving-state-of-the-art-medical","title":"Gazal-R1: Achieving State-of-the-Art Medical Reasoning with Parameter-Efficient Two-Stage Training","date":"2025-06-18","arxiv_id":"2506.21594","repositories_listed":0,"syntology":null},{"url":null,"slug":"slimming-down-llms-without-losing-their-minds","title":"Slimming Down LLMs Without Losing Their Minds","date":"2025-06-12","arxiv_id":"2506.10885","repositories_listed":0,"syntology":null},{"url":null,"slug":"surface-fairness-deep-bias-a-comparative","title":"Surface Fairness, Deep Bias: A Comparative Study of Bias in Language Models","date":"2025-06-12","arxiv_id":"2506.10491","repositories_listed":0,"syntology":null},{"url":null,"slug":"moe-gps-guidlines-for-prediction-strategy-for","title":"MoE-GPS: Guidlines for Prediction Strategy for Dynamic Expert Duplication in MoE Load Balancing","date":"2025-06-09","arxiv_id":"2506.07366","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10020","title":"From Threat to Tool: Leveraging Refusal-Aware Injection Attacks for Safety Alignment","date":"2025-06-07","arxiv_id":"2506.10020","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-robustness-stress-testing-of-llms","title":"Automatic Robustness Stress Testing of LLMs as Mathematical Problem Solvers","date":"2025-06-05","arxiv_id":"2506.05038","repositories_listed":0,"syntology":null},{"url":null,"slug":"gem-empowering-llm-for-both-embedding","title":"GEM: Empowering LLM for both Embedding Generation and Language Understanding","date":"2025-06-04","arxiv_id":"2506.04344","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-unlearning-via-sparse-autoencoder","title":"Model Unlearning via Sparse Autoencoder Subspace Guided Projections","date":"2025-05-30","arxiv_id":"2505.24428","repositories_listed":0,"syntology":null},{"url":"/paper/actor-critic-based-online-data-mixing-for","slug":"actor-critic-based-online-data-mixing-for","title":"Actor-Critic based Online Data Mixing For Language Model Pre-Training","date":"2025-05-29","arxiv_id":"2505.23878","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/actor-critic-based-online-data-mixing-for#ran","syntology_url":"https://syntology.ai/paper/2505.23878","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23878"}},"official":null}},{"url":null,"slug":"revisiting-uncertainty-estimation-and","title":"Revisiting Uncertainty Estimation and Calibration of Large Language Models","date":"2025-05-29","arxiv_id":"2505.23854","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-often-know-when-they","title":"Large Language Models Often Know When They Are Being Evaluated","date":"2025-05-28","arxiv_id":"2505.23836","repositories_listed":0,"syntology":null},{"url":null,"slug":"interleaved-reasoning-for-large-language","title":"Interleaved Reasoning for Large Language Models via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.19640","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-data-selection-at-scale-via","title":"Efficient Data Selection at Scale via Influence Distillation","date":"2025-05-25","arxiv_id":"2505.19051","repositories_listed":0,"syntology":null},{"url":null,"slug":"b-score-detecting-biases-in-large-language","title":"B-score: Detecting biases in large language models using response history","date":"2025-05-24","arxiv_id":"2505.18545","repositories_listed":0,"syntology":null},{"url":null,"slug":"inferencedynamics-efficient-routing-across","title":"INFERENCEDYNAMICS: Efficient Routing Across LLMs through Structured Capability and Knowledge Profiling","date":"2025-05-22","arxiv_id":"2505.16303","repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-aware-llm-based-online-dataset","title":"Cost-aware LLM-based Online Dataset Annotation","date":"2025-05-21","arxiv_id":"2505.15101","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-reasoner-incentivizing-reasoning","title":"Context Reasoner: Incentivizing Reasoning Capability for Contextualized Privacy and Safety Compliance via Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14585","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-decomposition-of-weights-and-singular","title":"Dual Decomposition of Weights and Singular Value Low Rank Adaptation","date":"2025-05-20","arxiv_id":"2505.14367","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-reasoning-language-models-unfold-hidden","title":"Self-Reasoning Language Models: Unfold Hidden Reasoning Chains with Few Reasoning Catalyst","date":"2025-05-20","arxiv_id":"2505.14116","repositories_listed":0,"syntology":null},{"url":null,"slug":"critique-guided-distillation-improving","title":"Critique-Guided Distillation: Improving Supervised Fine-tuning via Better Distillation","date":"2025-05-16","arxiv_id":"2505.11628","repositories_listed":0,"syntology":null},{"url":null,"slug":"mining-hidden-thoughts-from-texts-evaluating","title":"Mining Hidden Thoughts from Texts: Evaluating Continual Pretraining with Synthetic Data for LLM Reasoning","date":"2025-05-15","arxiv_id":"2505.10182","repositories_listed":0,"syntology":null},{"url":null,"slug":"kristeva-close-reading-as-a-novel-task-for","title":"KRISTEVA: Close Reading as a Novel Task for Benchmarking Interpretive Reasoning","date":"2025-05-14","arxiv_id":"2505.09825","repositories_listed":0,"syntology":null},{"url":null,"slug":"attentioninfluence-adopting-attention-head","title":"AttentionInfluence: Adopting Attention Head Influence for Weak-to-Strong Pretraining Data Selection","date":"2025-05-12","arxiv_id":"2505.07293","repositories_listed":0,"syntology":null},{"url":null,"slug":"sem-reinforcement-learning-for-search","title":"SEM: Reinforcement Learning for Search-Efficient Large Language Models","date":"2025-05-12","arxiv_id":"2505.07903","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-scaling-law-for-token-efficiency-in-llm","title":"A Scaling Law for Token Efficiency in LLM Fine-Tuning Under Fixed Compute Budgets","date":"2025-05-09","arxiv_id":"2505.06150","repositories_listed":0,"syntology":null},{"url":null,"slug":"elastic-weight-consolidation-for-full","title":"Elastic Weight Consolidation for Full-Parameter Continual Pre-Training of Gemma2","date":"2025-05-09","arxiv_id":"2505.05946","repositories_listed":0,"syntology":null},{"url":null,"slug":"llms-outperform-experts-on-challenging","title":"LLMs Outperform Experts on Challenging Biology Benchmarks","date":"2025-05-09","arxiv_id":"2505.06108","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-hong-kong-massive-multi-task","title":"Measuring Hong Kong Massive Multi-Task Language Understanding","date":"2025-05-04","arxiv_id":"2505.02177","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-efficient-llm-training-by-various","title":"Memory-Efficient LLM Training by Various-Grained Low-Rank Projection of Gradients","date":"2025-05-03","arxiv_id":"2505.01744","repositories_listed":0,"syntology":null},{"url":null,"slug":"longperceptualthoughts-distilling-system-2","title":"LongPerceptualThoughts: Distilling System-2 Reasoning for System-1 Perception","date":"2025-04-21","arxiv_id":"2504.15362","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferable-text-data-distillation-by","title":"Transferable text data distillation by trajectory matching","date":"2025-04-14","arxiv_id":"2504.09818","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptive-continued-pre-training-of","title":"Domain-Adaptive Continued Pre-Training of Small Language Models","date":"2025-04-13","arxiv_id":"2504.09687","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-could-be-rote-learners","title":"Large Language Models Could Be Rote Learners","date":"2025-04-11","arxiv_id":"2504.08300","repositories_listed":0,"syntology":null},{"url":null,"slug":"gaapo-genetic-algorithmic-applied-to-prompt","title":"GAAPO: Genetic Algorithmic Applied to Prompt Optimization","date":"2025-04-09","arxiv_id":"2504.07157","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-don-t-search-rethinking-test-time","title":"Sample, Don't Search: Rethinking Test-Time Alignment for Language Models","date":"2025-04-04","arxiv_id":"2504.03790","repositories_listed":0,"syntology":null},{"url":null,"slug":"more-is-less-the-pitfalls-of-multi-model","title":"More is Less: The Pitfalls of Multi-Model Synthetic Preference Data in DPO Safety Alignment","date":"2025-04-03","arxiv_id":"2504.02193","repositories_listed":0,"syntology":null},{"url":null,"slug":"order-independence-with-finetuning","title":"Order Independence With Finetuning","date":"2025-03-30","arxiv_id":"2503.23483","repositories_listed":0,"syntology":null}],"record_sha256":"8a700c3f6632633b545f219811a01566425b55f1ff11650568406f2ce58cf246","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}