{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/in-context-learning/papers/4","list_of":"/task/in-context-learning","task":"In-Context Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":23,"rows_per_page":100,"rows":[301,400],"of":2297,"counts":{"archive_papers_tagged":2297,"with_a_code_link":998,"where_syntology_ran_a_sample":437,"not_listed_spam_title":0,"listed":2297,"listed_where_code_ran":437,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":361,"every_run_a_failure_of_syntologys_instrument":76,"listed_with_a_run_with_no_instrument_failure":361,"listed_every_run_a_failure_of_syntologys_instrument":76,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/in-context-learning","prev":"/task/in-context-learning/papers/3","next":"/task/in-context-learning/papers/5","papers":[{"url":"/paper/aggregation-artifacts-in-subjective-tasks","slug":"aggregation-artifacts-in-subjective-tasks","title":"Aggregation Artifacts in Subjective Tasks Collapse Large Language Models' Posteriors","date":"2024-10-17","arxiv_id":"2410.13776","repositories_listed":1,"syntology":null},{"url":"/paper/bento-benchmark-task-reduction-with-in","slug":"bento-benchmark-task-reduction-with-in","title":"BenTo: Benchmark Task Reduction with In-Context Transferability","date":"2024-10-17","arxiv_id":"2410.13804","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bento-benchmark-task-reduction-with-in#ran","syntology_url":"https://syntology.ai/paper/2410.13804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13804"}},"official":{"repos":["tianyi-lab/bento"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/do-llms-overcome-shortcut-learning-an","slug":"do-llms-overcome-shortcut-learning-an","title":"Do LLMs Overcome Shortcut Learning? An Evaluation of Shortcut Challenges in Large Language Models","date":"2024-10-17","arxiv_id":"2410.13343","repositories_listed":1,"syntology":null},{"url":"/paper/in-context-learning-and-occam-s-razor","slug":"in-context-learning-and-occam-s-razor","title":"In-context learning and Occam's razor","date":"2024-10-17","arxiv_id":"2410.14086","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/in-context-learning-and-occam-s-razor#ran","syntology_url":"https://syntology.ai/paper/2410.14086","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14086"}},"official":{"repos":["3rdcore/prequentialcode"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-prompt-based-knowledge-graph-foundation","slug":"a-prompt-based-knowledge-graph-foundation","title":"A Prompt-Based Knowledge Graph Foundation Model for Universal In-Context Reasoning","date":"2024-10-16","arxiv_id":"2410.12288","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/a-prompt-based-knowledge-graph-foundation#ran","syntology_url":"https://syntology.ai/paper/2410.12288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12288"}},"official":{"repos":["nju-websoft/KG-ICL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/cognitive-overload-attack-prompt-injection","slug":"cognitive-overload-attack-prompt-injection","title":"Cognitive Overload Attack:Prompt Injection for Long Context","date":"2024-10-15","arxiv_id":"2410.11272","repositories_listed":1,"syntology":null},{"url":"/paper/rulerag-rule-guided-retrieval-augmented","slug":"rulerag-rule-guided-retrieval-augmented","title":"RuleRAG: Rule-guided retrieval-augmented generation with language models for question answering","date":"2024-10-15","arxiv_id":"2410.22353","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-model-based-reinforcement-learning","slug":"zero-shot-model-based-reinforcement-learning","title":"Zero-shot Model-based Reinforcement Learning using Large Language Models","date":"2024-10-15","arxiv_id":"2410.11711","repositories_listed":1,"syntology":null},{"url":"/paper/divide-reweight-and-conquer-a-logit","slug":"divide-reweight-and-conquer-a-logit","title":"Divide, Reweight, and Conquer: A Logit Arithmetic Approach for In-Context Learning","date":"2024-10-14","arxiv_id":"2410.10074","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/divide-reweight-and-conquer-a-logit#ran","syntology_url":"https://syntology.ai/paper/2410.10074","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10074"}},"official":{"repos":["chengsong-huang/lara"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kblam-knowledge-base-augmented-language-model","slug":"kblam-knowledge-base-augmented-language-model","title":"KBLaM: Knowledge Base augmented Language Model","date":"2024-10-14","arxiv_id":"2410.10450","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/kblam-knowledge-base-augmented-language-model#ran","syntology_url":"https://syntology.ai/paper/2410.10450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10450"}},"official":{"repos":["microsoft/KBLaM"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/will-llms-replace-the-encoder-only-models-in","slug":"will-llms-replace-the-encoder-only-models-in","title":"Will LLMs Replace the Encoder-Only Models in Temporal Relation Classification?","date":"2024-10-14","arxiv_id":"2410.10476","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/will-llms-replace-the-encoder-only-models-in#ran","syntology_url":"https://syntology.ai/paper/2410.10476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10476"}},"official":{"repos":["brownfortress/llms-trc"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/elicit-llm-augmentation-via-external-in","slug":"elicit-llm-augmentation-via-external-in","title":"ELICIT: LLM Augmentation via External In-Context Capability","date":"2024-10-12","arxiv_id":"2410.09343","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/elicit-llm-augmentation-via-external-in#ran","syntology_url":"https://syntology.ai/paper/2410.09343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09343"}},"official":{"repos":["lins-lab/elicit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/inference-and-verbalization-functions-during","slug":"inference-and-verbalization-functions-during","title":"Inference and Verbalization Functions During In-Context Learning","date":"2024-10-12","arxiv_id":"2410.09349","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inference-and-verbalization-functions-during#ran","syntology_url":"https://syntology.ai/paper/2410.09349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09349"}},"official":{"repos":["junyitao/infer-then-verbalize-during-icl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/appbench-planning-of-multiple-apis-from","slug":"appbench-planning-of-multiple-apis-from","title":"AppBench: Planning of Multiple APIs from Various APPs for Complex User Instruction","date":"2024-10-10","arxiv_id":"2410.19743","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/appbench-planning-of-multiple-apis-from#ran","syntology_url":"https://syntology.ai/paper/2410.19743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19743"}},"official":{"repos":["ruleGreen/AppBench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/metalic-meta-learning-in-context-with-protein","slug":"metalic-meta-learning-in-context-with-protein","title":"Metalic: Meta-Learning In-Context with Protein Language Models","date":"2024-10-10","arxiv_id":"2410.08355","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/metalic-meta-learning-in-context-with-protein#ran","syntology_url":"https://syntology.ai/paper/2410.08355","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08355"}},"official":{"repos":["instadeepai/metalic"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/plug-and-play-performance-estimation-for-llm","slug":"plug-and-play-performance-estimation-for-llm","title":"Plug-and-Play Performance Estimation for LLM Services without Relying on Labeled Data","date":"2024-10-10","arxiv_id":"2410.07737","repositories_listed":1,"syntology":null},{"url":"/paper/is-c4-dataset-optimal-for-pruning-an","slug":"is-c4-dataset-optimal-for-pruning-an","title":"Is C4 Dataset Optimal for Pruning? An Investigation of Calibration Data for LLM Pruning","date":"2024-10-09","arxiv_id":"2410.07461","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/is-c4-dataset-optimal-for-pruning-an#ran","syntology_url":"https://syntology.ai/paper/2410.07461","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07461"}},"official":{"repos":["abx393/llm-pruning-calibration-data"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/retrieval-augmented-decision-transformer","slug":"retrieval-augmented-decision-transformer","title":"Retrieval-Augmented Decision Transformer: External Memory for In-context RL","date":"2024-10-09","arxiv_id":"2410.07071","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/retrieval-augmented-decision-transformer#ran","syntology_url":"https://syntology.ai/paper/2410.07071","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07071"}},"official":{"repos":["ml-jku/RA-DT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/steering-large-language-models-using","slug":"steering-large-language-models-using","title":"Steering Large Language Models using Conceptors: Improving Addition-Based Activation Engineering","date":"2024-10-09","arxiv_id":"2410.16314","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/steering-large-language-models-using#ran","syntology_url":"https://syntology.ai/paper/2410.16314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16314"}},"official":{"repos":["jorispos/conceptorsteering"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/which-programming-language-and-what-features","slug":"which-programming-language-and-what-features","title":"Which Programming Language and What Features at Pre-training Stage Affect Downstream Logical Inference Performance?","date":"2024-10-09","arxiv_id":"2410.06735","repositories_listed":1,"syntology":null},{"url":"/paper/are-large-language-models-state-of-the-art","slug":"are-large-language-models-state-of-the-art","title":"Are Large Language Models State-of-the-art Quality Estimators for Machine Translation of User-generated Content?","date":"2024-10-08","arxiv_id":"2410.06338","repositories_listed":1,"syntology":null},{"url":"/paper/the-mystery-of-compositional-generalization","slug":"the-mystery-of-compositional-generalization","title":"The Mystery of Compositional Generalization in Graph-based Generative Commonsense Reasoning","date":"2024-10-08","arxiv_id":"2410.06272","repositories_listed":1,"syntology":null},{"url":"/paper/vector-icl-in-context-learning-with","slug":"vector-icl-in-context-learning-with","title":"Vector-ICL: In-context Learning with Continuous Vector Representations","date":"2024-10-08","arxiv_id":"2410.05629","repositories_listed":1,"syntology":null},{"url":"/paper/a-simple-image-segmentation-framework-via-in","slug":"a-simple-image-segmentation-framework-via-in","title":"A Simple Image Segmentation Framework via In-Context Examples","date":"2024-10-07","arxiv_id":"2410.04842","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-simple-image-segmentation-framework-via-in#ran","syntology_url":"https://syntology.ai/paper/2410.04842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04842"}},"official":{"repos":["aim-uofa/sine"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deeper-insights-without-updates-the-power-of","slug":"deeper-insights-without-updates-the-power-of","title":"Deeper Insights Without Updates: The Power of In-Context Learning Over Fine-Tuning","date":"2024-10-07","arxiv_id":"2410.04691","repositories_listed":1,"syntology":null},{"url":"/paper/task-diversity-shortens-the-icl-plateau","slug":"task-diversity-shortens-the-icl-plateau","title":"Task Diversity Shortens the ICL Plateau","date":"2024-10-07","arxiv_id":"2410.05448","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/task-diversity-shortens-the-icl-plateau#ran","syntology_url":"https://syntology.ai/paper/2410.05448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05448"}},"official":{"repos":["sehyunkwon/task-diversity-icl"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-in-context-learning-inference","slug":"revisiting-in-context-learning-inference","title":"Revisiting In-context Learning Inference Circuit in Large Language Models","date":"2024-10-06","arxiv_id":"2410.04468","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/revisiting-in-context-learning-inference#ran","syntology_url":"https://syntology.ai/paper/2410.04468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04468"}},"official":{"repos":["hc495/ICL_Circuit"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-large-language-models-for-inverse","slug":"multimodal-large-language-models-for-inverse","title":"Multimodal Large Language Models for Inverse Molecular Design with Retrosynthetic Planning","date":"2024-10-05","arxiv_id":"2410.04223","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multimodal-large-language-models-for-inverse#ran","syntology_url":"https://syntology.ai/paper/2410.04223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04223"}},"official":{"repos":["liugangcode/Llamole"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/generating-bilingual-example-sentences-with","slug":"generating-bilingual-example-sentences-with","title":"Generating bilingual example sentences with large language models as lexicography assistants","date":"2024-10-04","arxiv_id":"2410.03182","repositories_listed":1,"syntology":null},{"url":"/paper/in-context-learning-in-presence-of-spurious","slug":"in-context-learning-in-presence-of-spurious","title":"In-context Learning in Presence of Spurious Correlations","date":"2024-10-04","arxiv_id":"2410.03140","repositories_listed":1,"syntology":null},{"url":"/paper/personalsum-a-user-subjective-guided","slug":"personalsum-a-user-subjective-guided","title":"PersonalSum: A User-Subjective Guided Personalized Summarization Dataset for Large Language Models","date":"2024-10-04","arxiv_id":"2410.03905","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/personalsum-a-user-subjective-guided#ran","syntology_url":"https://syntology.ai/paper/2410.03905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03905"}},"official":{"repos":["smartmediaai/personalsum"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ripplecot-amplifying-ripple-effect-of","slug":"ripplecot-amplifying-ripple-effect-of","title":"RIPPLECOT: Amplifying Ripple Effect of Knowledge Editing in Language Models via Chain-of-Thought In-Context Learning","date":"2024-10-04","arxiv_id":"2410.03122","repositories_listed":1,"syntology":null},{"url":"/paper/relic-a-recipe-for-64k-steps-of-in-context","slug":"relic-a-recipe-for-64k-steps-of-in-context","title":"ReLIC: A Recipe for 64k Steps of In-Context Reinforcement Learning for Embodied AI","date":"2024-10-03","arxiv_id":"2410.02751","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/relic-a-recipe-for-64k-steps-of-in-context#ran","syntology_url":"https://syntology.ai/paper/2410.02751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02751"}},"official":{"repos":["aielawady/relic"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unleashing-the-potential-of-the-diffusion","slug":"unleashing-the-potential-of-the-diffusion","title":"Unleashing the Potential of the Diffusion Model in Few-shot Semantic Segmentation","date":"2024-10-03","arxiv_id":"2410.02369","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unleashing-the-potential-of-the-diffusion#ran","syntology_url":"https://syntology.ai/paper/2410.02369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02369"}},"official":{"repos":["aim-uofa/diffews"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bayes-power-for-explaining-in-context","slug":"bayes-power-for-explaining-in-context","title":"Bayes' Power for Explaining In-Context Learning Generalizations","date":"2024-10-02","arxiv_id":"2410.01565","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bayes-power-for-explaining-in-context#ran","syntology_url":"https://syntology.ai/paper/2410.01565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01565"}},"official":{"repos":["samuelgabriel/bayesgeneralizations"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/in-context-transfer-learning-demonstration","slug":"in-context-transfer-learning-demonstration","title":"In-Context Transfer Learning: Demonstration Synthesis by Transferring Similar Tasks","date":"2024-10-02","arxiv_id":"2410.01548","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-language-skills-under-circuits","slug":"unveiling-language-skills-under-circuits","title":"Unveiling Language Skills via Path-Level Circuit Discovery","date":"2024-10-02","arxiv_id":"2410.01334","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unveiling-language-skills-under-circuits#ran","syntology_url":"https://syntology.ai/paper/2410.01334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01334"}},"official":{"repos":["zodiark-ch/language-skill-of-llms"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/personalllm-tailoring-llms-to-individual","slug":"personalllm-tailoring-llms-to-individual","title":"PersonalLLM: Tailoring LLMs to Individual Preferences","date":"2024-09-30","arxiv_id":"2409.20296","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/personalllm-tailoring-llms-to-individual#ran","syntology_url":"https://syntology.ai/paper/2409.20296","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20296"}},"official":{"repos":["namkoong-lab/PersonalLLM"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/reference-trustable-decoding-a-training-free","slug":"reference-trustable-decoding-a-training-free","title":"Reference Trustable Decoding: A Training-Free Augmentation Paradigm for Large Language Models","date":"2024-09-30","arxiv_id":"2409.20181","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/reference-trustable-decoding-a-training-free#ran","syntology_url":"https://syntology.ai/paper/2409.20181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20181"}},"official":{"repos":["shiluohe/referencetrustabledecoding"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/taskcomplexity-a-dataset-for-task-complexity","slug":"taskcomplexity-a-dataset-for-task-complexity","title":"TaskComplexity: A Dataset for Task Complexity Classification with In-Context Learning, FLAN-T5 and GPT-4o Benchmarks","date":"2024-09-30","arxiv_id":"2409.20189","repositories_listed":1,"syntology":null},{"url":"/paper/text-clustering-as-classification-with-llms","slug":"text-clustering-as-classification-with-llms","title":"Text Clustering as Classification with LLMs","date":"2024-09-30","arxiv_id":"2410.00927","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/text-clustering-as-classification-with-llms#ran","syntology_url":"https://syntology.ai/paper/2410.00927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00927"}},"official":{"repos":["ecnu-text-computing/text-clustering-via-llm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/t2vs-meet-vlms-a-scalable-multimodal-dataset","slug":"t2vs-meet-vlms-a-scalable-multimodal-dataset","title":"T2Vs Meet VLMs: A Scalable Multimodal Dataset for Visual Harmfulness Recognition","date":"2024-09-29","arxiv_id":"2409.19734","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/t2vs-meet-vlms-a-scalable-multimodal-dataset#ran","syntology_url":"https://syntology.ai/paper/2409.19734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.19734"}},"official":{"repos":["nctu-eva-lab/vhd11k"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/pioneering-reliable-assessment-in-text-to","slug":"pioneering-reliable-assessment-in-text-to","title":"Pioneering Reliable Assessment in Text-to-Image Knowledge Editing: Leveraging a Fine-Grained Dataset and an Innovative Criterion","date":"2024-09-26","arxiv_id":"2409.17928","repositories_listed":1,"syntology":null},{"url":"/paper/can-vision-language-models-learn-from-visual","slug":"can-vision-language-models-learn-from-visual","title":"Can Vision Language Models Learn from Visual Demonstrations of Ambiguous Spatial Reasoning?","date":"2024-09-25","arxiv_id":"2409.17080","repositories_listed":1,"syntology":null},{"url":"/paper/in-context-ensemble-improves-video-language","slug":"in-context-ensemble-improves-video-language","title":"In-Context Ensemble Learning from Pseudo Labels Improves Video-Language Models for Low-Level Workflow Understanding","date":"2024-09-24","arxiv_id":"2409.15867","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/in-context-ensemble-improves-video-language#ran","syntology_url":"https://syntology.ai/paper/2409.15867","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15867"}},"official":{"repos":["moucheng2017/action-labelling"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/making-text-embedders-few-shot-learners","slug":"making-text-embedders-few-shot-learners","title":"Making Text Embedders Few-Shot Learners","date":"2024-09-24","arxiv_id":"2409.15700","repositories_listed":1,"syntology":null},{"url":"/paper/small-language-models-survey-measurements-and","slug":"small-language-models-survey-measurements-and","title":"Small Language Models: Survey, Measurements, and Insights","date":"2024-09-24","arxiv_id":"2409.15790","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-forecasting-of-chaotic-systems","slug":"zero-shot-forecasting-of-chaotic-systems","title":"Zero-shot forecasting of chaotic systems","date":"2024-09-24","arxiv_id":"2409.15771","repositories_listed":1,"syntology":{"n":20,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/zero-shot-forecasting-of-chaotic-systems#ran","syntology_url":"https://syntology.ai/paper/2409.15771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15771"}},"official":{"repos":["williamgilpin/dysts"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/in-context-learning-may-not-elicit","slug":"in-context-learning-may-not-elicit","title":"In-Context Learning May Not Elicit Trustworthy Reasoning: A-Not-B Errors in Pretrained Language Models","date":"2024-09-23","arxiv_id":"2409.15454","repositories_listed":1,"syntology":null},{"url":"/paper/inferring-scientific-cross-document","slug":"inferring-scientific-cross-document","title":"Inferring Scientific Cross-Document Coreference and Hierarchy with Definition-Augmented Relational Reasoning","date":"2024-09-23","arxiv_id":"2409.15113","repositories_listed":1,"syntology":null},{"url":"/paper/pallm-evaluating-and-enhancing-palliative","slug":"pallm-evaluating-and-enhancing-palliative","title":"PALLM: Evaluating and Enhancing PALLiative Care Conversations with Large Language Models","date":"2024-09-23","arxiv_id":"2409.15188","repositories_listed":1,"syntology":null},{"url":"/paper/revise-reason-and-recognize-llm-based-emotion","slug":"revise-reason-and-recognize-llm-based-emotion","title":"Revise, Reason, and Recognize: LLM-Based Emotion Recognition via Emotion-Specific Prompts and ASR Error Correction","date":"2024-09-23","arxiv_id":"2409.15551","repositories_listed":1,"syntology":null},{"url":"/paper/one-shot-world-models-using-a-transformer","slug":"one-shot-world-models-using-a-transformer","title":"One-shot World Models Using a Transformer Trained on a Synthetic Prior","date":"2024-09-21","arxiv_id":"2409.14084","repositories_listed":1,"syntology":null},{"url":"/paper/stateact-state-tracking-and-reasoning-for","slug":"stateact-state-tracking-and-reasoning-for","title":"StateAct: State Tracking and Reasoning for Acting and Planning with Large Language Models","date":"2024-09-21","arxiv_id":"2410.02810","repositories_listed":1,"syntology":{"n":19,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":19,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/stateact-state-tracking-and-reasoning-for#ran","syntology_url":"https://syntology.ai/paper/2410.02810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02810"}},"official":{"repos":["ai-nikolai/stateact"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/a-controlled-study-on-long-context-extension","slug":"a-controlled-study-on-long-context-extension","title":"A Controlled Study on Long Context Extension and Generalization in LLMs","date":"2024-09-18","arxiv_id":"2409.12181","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":5,"n_pointer_only":14,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-controlled-study-on-long-context-extension#ran","syntology_url":"https://syntology.ai/paper/2409.12181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12181"}},"official":{"repos":["leooyii/lceg"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ruie-retrieval-based-unified-information","slug":"ruie-retrieval-based-unified-information","title":"RUIE: Retrieval-based Unified Information Extraction using Large Language Model","date":"2024-09-18","arxiv_id":"2409.11673","repositories_listed":1,"syntology":null},{"url":"/paper/hearts-a-holistic-framework-for-explainable","slug":"hearts-a-holistic-framework-for-explainable","title":"HEARTS: A Holistic Framework for Explainable, Sustainable and Robust Text Stereotype Detection","date":"2024-09-17","arxiv_id":"2409.11579","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-and-enhancing-trustworthiness-of","slug":"measuring-and-enhancing-trustworthiness-of","title":"Measuring and Enhancing Trustworthiness of LLMs in RAG through Grounded Attributions and Learning to Refuse","date":"2024-09-17","arxiv_id":"2409.11242","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/measuring-and-enhancing-trustworthiness-of#ran","syntology_url":"https://syntology.ai/paper/2409.11242","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.11242"}},"official":{"repos":["declare-lab/trust-align"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reasoning-graph-enhanced-exemplars-retrieval","slug":"reasoning-graph-enhanced-exemplars-retrieval","title":"Reasoning Graph Enhanced Exemplars Retrieval for In-Context Learning","date":"2024-09-17","arxiv_id":"2409.11147","repositories_listed":1,"syntology":null},{"url":"/paper/thames-an-end-to-end-tool-for-hallucination","slug":"thames-an-end-to-end-tool-for-hallucination","title":"THaMES: An End-to-End Tool for Hallucination Mitigation and Evaluation in Large Language Models","date":"2024-09-17","arxiv_id":"2409.11353","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/thames-an-end-to-end-tool-for-hallucination#ran","syntology_url":"https://syntology.ai/paper/2409.11353","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.11353"}},"official":{"repos":["holistic-ai/THaMES"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/do-large-language-models-need-a-content","slug":"do-large-language-models-need-a-content","title":"Do Large Language Models Need a Content Delivery Network?","date":"2024-09-16","arxiv_id":"2409.13761","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/do-large-language-models-need-a-content#ran","syntology_url":"https://syntology.ai/paper/2409.13761","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.13761"}},"official":{"repos":["lmcache/lmcache"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/select-sql-self-correcting-ensemble-chain-of","slug":"select-sql-self-correcting-ensemble-chain-of","title":"SelECT-SQL: Self-correcting ensemble Chain-of-Thought for Text-to-SQL","date":"2024-09-16","arxiv_id":"2409.10007","repositories_listed":1,"syntology":null},{"url":"/paper/alpapico-extraction-of-pico-frames-from","slug":"alpapico-extraction-of-pico-frames-from","title":"AlpaPICO: Extraction of PICO Frames from Clinical Trial Documents Using LLMs","date":"2024-09-15","arxiv_id":"2409.09704","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-rare-word-accuracy-in-direct","slug":"optimizing-rare-word-accuracy-in-direct","title":"Optimizing Rare Word Accuracy in Direct Speech Translation with a Retrieval-and-Demonstration Approach","date":"2024-09-13","arxiv_id":"2409.09009","repositories_listed":1,"syntology":null},{"url":"/paper/fine-tuning-large-language-models-for-entity","slug":"fine-tuning-large-language-models-for-entity","title":"Fine-tuning Large Language Models for Entity Matching","date":"2024-09-12","arxiv_id":"2409.08185","repositories_listed":1,"syntology":null},{"url":"/paper/larger-language-models-don-t-care-how-you","slug":"larger-language-models-don-t-care-how-you","title":"Larger Language Models Don't Care How You Think: Why Chain-of-Thought Prompting Fails in Subjective Tasks","date":"2024-09-10","arxiv_id":"2409.06173","repositories_listed":1,"syntology":null},{"url":"/paper/2409-13728","slug":"2409-13728","title":"Rule Extrapolation in Language Models: A Study of Compositional Generalization on OOD Prompts","date":"2024-09-09","arxiv_id":"2409.13728","repositories_listed":1,"syntology":null},{"url":"/paper/mile-a-mutation-testing-framework-of-in","slug":"mile-a-mutation-testing-framework-of-in","title":"MILE: A Mutation Testing Framework of In-Context Learning Systems","date":"2024-09-07","arxiv_id":"2409.04831","repositories_listed":1,"syntology":null},{"url":"/paper/learning-vs-retrieval-the-role-of-in-context","slug":"learning-vs-retrieval-the-role-of-in-context","title":"Learning vs Retrieval: The Role of In-Context Examples in Regression with LLMs","date":"2024-09-06","arxiv_id":"2409.04318","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-vs-retrieval-the-role-of-in-context#ran","syntology_url":"https://syntology.ai/paper/2409.04318","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.04318"}},"official":{"repos":["HLR/LvsR-LLM"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cacer-clinical-concept-annotations-for-cancer","slug":"cacer-clinical-concept-annotations-for-cancer","title":"CACER: Clinical Concept Annotations for Cancer Events and Relations","date":"2024-09-05","arxiv_id":"2409.03905","repositories_listed":1,"syntology":null},{"url":"/paper/the-representation-landscape-of-few-shot","slug":"the-representation-landscape-of-few-shot","title":"The representation landscape of few-shot learning and fine-tuning in large language models","date":"2024-09-05","arxiv_id":"2409.03662","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-representation-landscape-of-few-shot#ran","syntology_url":"https://syntology.ai/paper/2409.03662","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.03662"}},"official":{"repos":["diegodoimo/geometry_icl_finetuning"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-determine-the-preferred-image","slug":"how-to-determine-the-preferred-image","title":"How to Determine the Preferred Image Distribution of a Black-Box Vision-Language Model?","date":"2024-09-03","arxiv_id":"2409.02253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-determine-the-preferred-image#ran","syntology_url":"https://syntology.ai/paper/2409.02253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02253"}},"official":{"repos":["asgsaeid/cad_vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-compressor-retriever-architecture-for","slug":"the-compressor-retriever-architecture-for","title":"The Compressor-Retriever Architecture for Language Model OS","date":"2024-09-02","arxiv_id":"2409.01495","repositories_listed":1,"syntology":null},{"url":"/paper/self-alignment-improving-alignment-of","slug":"self-alignment-improving-alignment-of","title":"Self-Alignment: Improving Alignment of Cultural Values in LLMs via In-Context Learning","date":"2024-08-29","arxiv_id":"2408.16482","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-named-entity-recognition-using-few","slug":"evaluating-named-entity-recognition-using-few","title":"Evaluating Named Entity Recognition Using Few-Shot Prompting with Large Language Models","date":"2024-08-28","arxiv_id":"2408.15796","repositories_listed":1,"syntology":null},{"url":"/paper/foundation-models-for-music-a-survey","slug":"foundation-models-for-music-a-survey","title":"Foundation Models for Music: A Survey","date":"2024-08-26","arxiv_id":"2408.14340","repositories_listed":1,"syntology":null},{"url":"/paper/causal-guided-active-learning-for-debiasing","slug":"causal-guided-active-learning-for-debiasing","title":"Causal-Guided Active Learning for Debiasing Large Language Models","date":"2024-08-23","arxiv_id":"2408.12942","repositories_listed":1,"syntology":null},{"url":"/paper/evidence-backed-fact-checking-using-rag-and","slug":"evidence-backed-fact-checking-using-rag-and","title":"Evidence-backed Fact Checking using RAG and Few-Shot In-Context Learning with LLMs","date":"2024-08-22","arxiv_id":"2408.12060","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-language-models-for-math","slug":"benchmarking-large-language-models-for-math","title":"Benchmarking Large Language Models for Math Reasoning Tasks","date":"2024-08-20","arxiv_id":"2408.10839","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-verilogeval-newer-llms-in-context","slug":"revisiting-verilogeval-newer-llms-in-context","title":"Revisiting VerilogEval: A Year of Improvements in Large-Language Models for Hardware Code Generation","date":"2024-08-20","arxiv_id":"2408.11053","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/revisiting-verilogeval-newer-llms-in-context#ran","syntology_url":"https://syntology.ai/paper/2408.11053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11053"}},"official":{"repos":["nvlabs/verilog-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/v-roast-a-new-dataset-for-visual-road","slug":"v-roast-a-new-dataset-for-visual-road","title":"V-RoAst: Visual Road Assessment. Can VLM be a Road Safety Assessor Using the iRAP Standard?","date":"2024-08-20","arxiv_id":"2408.10872","repositories_listed":1,"syntology":null},{"url":"/paper/out-of-distribution-generalization-via-1","slug":"out-of-distribution-generalization-via-1","title":"Out-of-distribution generalization via composition: a lens through induction heads in Transformers","date":"2024-08-18","arxiv_id":"2408.09503","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/out-of-distribution-generalization-via-1#ran","syntology_url":"https://syntology.ai/paper/2408.09503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09503"}},"official":{"repos":["jiajunsong629/ood-generalization-via-composition"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/xgen-mm-blip-3-a-family-of-open-large","slug":"xgen-mm-blip-3-a-family-of-open-large","title":"xGen-MM (BLIP-3): A Family of Open Large Multimodal Models","date":"2024-08-16","arxiv_id":"2408.08872","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/xgen-mm-blip-3-a-family-of-open-large#ran","syntology_url":"https://syntology.ai/paper/2408.08872","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08872"}},"official":null}},{"url":"/paper/arablegaleval-a-multitask-benchmark-for","slug":"arablegaleval-a-multitask-benchmark-for","title":"ArabLegalEval: A Multitask Benchmark for Assessing Arabic Legal Knowledge in Large Language Models","date":"2024-08-15","arxiv_id":"2408.07983","repositories_listed":1,"syntology":null},{"url":"/paper/mag-sql-multi-agent-generative-approach-with","slug":"mag-sql-multi-agent-generative-approach-with","title":"MAG-SQL: Multi-Agent Generative Approach with Soft Schema Linking and Iterative Sub-SQL Refinement for Text-to-SQL","date":"2024-08-15","arxiv_id":"2408.07930","repositories_listed":1,"syntology":{"n":25,"n_ran":21,"n_constructed":0,"n_ran_checked":19,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":3,"n_no_contract":15,"n_pointer_only":4,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 1 honoured, 3 violated, 15 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mag-sql-multi-agent-generative-approach-with#ran","syntology_url":"https://syntology.ai/paper/2408.07930","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07930"}},"official":{"repos":["LancelotXWX/MAG-SQL"],"state":"official (archive's flag): 21 ran","n_ran":21,"n_constructed":0,"n_ran_no_instrument_failure":19,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-autonomous-agents-adaptive-planning","slug":"towards-autonomous-agents-adaptive-planning","title":"Towards Autonomous Agents: Adaptive-planning, Reasoning, and Acting in Language Models","date":"2024-08-12","arxiv_id":"2408.06458","repositories_listed":1,"syntology":null},{"url":"/paper/laida-linguistics-aware-in-context-learning","slug":"laida-linguistics-aware-in-context-learning","title":"LaiDA: Linguistics-aware In-context Learning with Data Augmentation for Metaphor Components Identification","date":"2024-08-10","arxiv_id":"2408.05404","repositories_listed":1,"syntology":null},{"url":"/paper/scoi-syntax-augmented-coverage-based-in","slug":"scoi-syntax-augmented-coverage-based-in","title":"SCOI: Syntax-augmented Coverage-based In-context Example Selection for Machine Translation","date":"2024-08-09","arxiv_id":"2408.04872","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/scoi-syntax-augmented-coverage-based-in#ran","syntology_url":"https://syntology.ai/paper/2408.04872","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04872"}},"official":{"repos":["jamydon/scoi"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/learning-fine-grained-grounded-citations-for","slug":"learning-fine-grained-grounded-citations-for","title":"Learning Fine-Grained Grounded Citations for Attributed Large Language Models","date":"2024-08-08","arxiv_id":"2408.04568","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-fine-grained-grounded-citations-for#ran","syntology_url":"https://syntology.ai/paper/2408.04568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04568"}},"official":{"repos":["luckyyysta/fine-grained-attribution"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/optimus-1-hybrid-multimodal-memory-empowered","slug":"optimus-1-hybrid-multimodal-memory-empowered","title":"Optimus-1: Hybrid Multimodal Memory Empowered Agents Excel in Long-Horizon Tasks","date":"2024-08-07","arxiv_id":"2408.03615","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/optimus-1-hybrid-multimodal-memory-empowered#ran","syntology_url":"https://syntology.ai/paper/2408.03615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03615"}},"official":{"repos":["JiuTian-VL/Optimus-1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00994","slug":"2408-00994","title":"ArchCode: Incorporating Software Requirements in Code Generation with Large Language Models","date":"2024-08-02","arxiv_id":"2408.00994","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01088","slug":"2408-01088","title":"Bridging Information Gaps in Dialogues With Grounded Exchanges Using Knowledge Graphs","date":"2024-08-02","arxiv_id":"2408.01088","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01585","slug":"2408-01585","title":"LibreLog: Accurate and Efficient Unsupervised Log Parsing Using Open-Source Large Language Models","date":"2024-08-02","arxiv_id":"2408.01585","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00274","slug":"2408-00274","title":"QUITO: Accelerating Long-Context Reasoning through Query-Guided Context Compression","date":"2024-08-01","arxiv_id":"2408.00274","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00312","slug":"2408-00312","title":"Adversarial Text Rewriting for Text-aware Recommender Systems","date":"2024-08-01","arxiv_id":"2408.00312","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00397","slug":"2408-00397","title":"In-Context Example Selection via Similarity Search Improves Low-Resource Machine Translation","date":"2024-08-01","arxiv_id":"2408.00397","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00523","slug":"2408-00523","title":"Fuzz-Testing Meets LLM-Based Agents: An Automated and Efficient Framework for Jailbreaking Text-To-Image Generation Models","date":"2024-08-01","arxiv_id":"2408.00523","repositories_listed":1,"syntology":null},{"url":"/paper/polynomial-regression-as-a-task-for","slug":"polynomial-regression-as-a-task-for","title":"Polynomial Regression as a Task for Understanding In-context Learning Through Finetuning and Alignment","date":"2024-07-27","arxiv_id":"2407.19346","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":2,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/polynomial-regression-as-a-task-for#ran","syntology_url":"https://syntology.ai/paper/2407.19346","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.19346"}},"official":{"repos":["MSNetrom/in-context-poly-playground"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/grammar-based-game-description-generation","slug":"grammar-based-game-description-generation","title":"Grammar-based Game Description Generation using Large Language Models","date":"2024-07-24","arxiv_id":"2407.17404","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-for-anomaly-detection","slug":"large-language-models-for-anomaly-detection","title":"Large Language Models for Anomaly Detection in Computational Workflows: from Supervised Fine-Tuning to In-Context Learning","date":"2024-07-24","arxiv_id":"2407.17545","repositories_listed":1,"syntology":null}],"record_sha256":"050ab33228b1d13b6d5357853aa25c08a4c310dfc03d487289d1a1781b6b6aac","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}