{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/prompt-engineering/papers/ran/1","list_of":"/task/prompt-engineering","task":"Prompt Engineering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":143,"counts":{"archive_papers_tagged":1236,"with_a_code_link":454,"where_syntology_ran_a_sample":143,"not_listed_spam_title":0,"listed":1236,"listed_where_code_ran":143,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":118,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":118,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/prompt-engineering/papers/ran/1","prev":null,"next":"/task/prompt-engineering/papers/ran/2","papers":[{"url":"/paper/2506-08184","slug":"2506-08184","title":"Unable to Forget: Proactive lnterference Reveals Working Memory Limits in LLMs Beyond Context Length","date":"2025-06-09","arxiv_id":"2506.08184","repositories_listed":0,"syntology":{"n":26,"n_ran":25,"n_constructed":0,"n_ran_checked":25,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":25,"n_pointer_only":0,"phrase":"25 ran (of which 0 constructed an object rather than computing a result; 25 with no instrument failure: 0 honoured, 0 violated, 25 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2506-08184#ran","syntology_url":"https://syntology.ai/paper/2506.08184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08184"}},"official":null}},{"url":"/paper/breaking-the-ceiling-exploring-the-potential","slug":"breaking-the-ceiling-exploring-the-potential","title":"Breaking the Ceiling: Exploring the Potential of Jailbreak Attacks through Expanding Strategy Space","date":"2025-05-27","arxiv_id":"2505.21277","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/breaking-the-ceiling-exploring-the-potential#ran","syntology_url":"https://syntology.ai/paper/2505.21277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21277"}},"official":{"repos":["aries-iai/cl-gso"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/capability-based-scaling-laws-for-llm-red","slug":"capability-based-scaling-laws-for-llm-red","title":"Capability-Based Scaling Laws for LLM Red-Teaming","date":"2025-05-26","arxiv_id":"2505.20162","repositories_listed":1,"syntology":{"n":23,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":0,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/capability-based-scaling-laws-for-llm-red#ran","syntology_url":"https://syntology.ai/paper/2505.20162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20162"}},"official":{"repos":["kotekjedi/capability-based-scaling"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":17,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-prompt-engineering-robust-behavior","slug":"beyond-prompt-engineering-robust-behavior","title":"Beyond Prompt Engineering: Robust Behavior Control in LLMs via Steering Target Atoms","date":"2025-05-23","arxiv_id":"2505.20322","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-prompt-engineering-robust-behavior#ran","syntology_url":"https://syntology.ai/paper/2505.20322","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20322"}},"official":{"repos":["zjunlp/steer-target-atoms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/concept-level-explainability-for-auditing","slug":"concept-level-explainability-for-auditing","title":"Concept-Level Explainability for Auditing & Steering LLM Responses","date":"2025-05-12","arxiv_id":"2505.07610","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/concept-level-explainability-for-auditing#ran","syntology_url":"https://syntology.ai/paper/2505.07610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07610"}},"official":{"repos":["k-amara/ConceptX"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-evaluative-thinking-meta-policy","slug":"toward-evaluative-thinking-meta-policy","title":"Toward Evaluative Thinking: Meta Policy Optimization with Evolving Reward Models","date":"2025-04-28","arxiv_id":"2504.20157","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":15,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/toward-evaluative-thinking-meta-policy#ran","syntology_url":"https://syntology.ai/paper/2504.20157","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20157"}},"official":{"repos":["minnesotanlp/mpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-on-prompt-compression-for","slug":"an-empirical-study-on-prompt-compression-for","title":"An Empirical Study on Prompt Compression for Large Language Models","date":"2025-04-24","arxiv_id":"2505.00019","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-on-prompt-compression-for#ran","syntology_url":"https://syntology.ai/paper/2505.00019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.00019"}},"official":{"repos":["3DAgentWorld/Toolkit-for-Prompt-Compression"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/private-federated-learning-using-preference","slug":"private-federated-learning-using-preference","title":"Private Federated Learning using Preference-Optimized Synthetic Data","date":"2025-04-23","arxiv_id":"2504.16438","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/private-federated-learning-using-preference#ran","syntology_url":"https://syntology.ai/paper/2504.16438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16438"}},"official":{"repos":["meiyuw/popri"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deepresearcher-scaling-deep-research-via","slug":"deepresearcher-scaling-deep-research-via","title":"DeepResearcher: Scaling Deep Research via Reinforcement Learning in Real-world Environments","date":"2025-04-04","arxiv_id":"2504.03160","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepresearcher-scaling-deep-research-via#ran","syntology_url":"https://syntology.ai/paper/2504.03160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.03160"}},"official":{"repos":["gair-nlp/deepresearcher"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-vision-language-models-are-unsupervised","slug":"large-vision-language-models-are-unsupervised","title":"Large (Vision) Language Models are Unsupervised In-Context Learners","date":"2025-04-03","arxiv_id":"2504.02349","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/large-vision-language-models-are-unsupervised#ran","syntology_url":"https://syntology.ai/paper/2504.02349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02349"}},"official":{"repos":["mlbio-epfl/joint-inference"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-reasoning-to-adapt-large-language","slug":"enhancing-reasoning-to-adapt-large-language","title":"Enhancing Reasoning to Adapt Large Language Models for Domain-Specific Applications","date":"2025-02-05","arxiv_id":"2502.04384","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/enhancing-reasoning-to-adapt-large-language#ran","syntology_url":"https://syntology.ai/paper/2502.04384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04384"}},"official":{"repos":["wenboown/generative-ai-for-semiconductor-physical-design"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-modal-prototype-joint-learning-for","slug":"dual-modal-prototype-joint-learning-for","title":"Dual-Modal Prototype Joint Learning for Compositional Zero-Shot Learning","date":"2025-01-23","arxiv_id":"2501.13859","repositories_listed":0,"syntology":{"n":8,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/dual-modal-prototype-joint-learning-for#ran","syntology_url":"https://syntology.ai/paper/2501.13859","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.13859"}},"official":null}},{"url":"/paper/greater-gradients-over-reasoning-makes","slug":"greater-gradients-over-reasoning-makes","title":"GReaTer: Gradients over Reasoning Makes Smaller Language Models Strong Prompt Optimizers","date":"2024-12-12","arxiv_id":"2412.09722","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/greater-gradients-over-reasoning-makes#ran","syntology_url":"https://syntology.ai/paper/2412.09722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09722"}},"official":{"repos":["psunlpgroup/greater"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/3d-part-segmentation-via-geometric","slug":"3d-part-segmentation-via-geometric","title":"3D Part Segmentation via Geometric Aggregation of 2D Visual Features","date":"2024-12-05","arxiv_id":"2412.04247","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3d-part-segmentation-via-geometric#ran","syntology_url":"https://syntology.ai/paper/2412.04247","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.04247"}},"official":{"repos":["marco-garosi/COPS"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revolve-optimizing-ai-systems-by-tracking","slug":"revolve-optimizing-ai-systems-by-tracking","title":"Revolve: Optimizing AI Systems by Tracking Response Evolution in Textual Optimization","date":"2024-12-04","arxiv_id":"2412.03092","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/revolve-optimizing-ai-systems-by-tracking#ran","syntology_url":"https://syntology.ai/paper/2412.03092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03092"}},"official":{"repos":["peiyance/revolve"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/from-text-to-pose-to-image-improving","slug":"from-text-to-pose-to-image-improving","title":"From Text to Pose to Image: Improving Diffusion Model Control and Quality","date":"2024-11-19","arxiv_id":"2411.12872","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":3,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/from-text-to-pose-to-image-improving#ran","syntology_url":"https://syntology.ai/paper/2411.12872","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.12872"}},"official":{"repos":["clement-bonnet/text-to-pose"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":3,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sg-bench-evaluating-llm-safety-generalization","slug":"sg-bench-evaluating-llm-safety-generalization","title":"SG-Bench: Evaluating LLM Safety Generalization Across Diverse Tasks and Prompt Types","date":"2024-10-29","arxiv_id":"2410.21965","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sg-bench-evaluating-llm-safety-generalization#ran","syntology_url":"https://syntology.ai/paper/2410.21965","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21965"}},"official":{"repos":["MurrayTom/SG-Bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/do-llms-know-internally-when-they-follow","slug":"do-llms-know-internally-when-they-follow","title":"Do LLMs \"know\" internally when they follow instructions?","date":"2024-10-18","arxiv_id":"2410.14516","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/do-llms-know-internally-when-they-follow#ran","syntology_url":"https://syntology.ai/paper/2410.14516","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14516"}},"official":{"repos":["apple/ml-internal-llms-instruction-following"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/self-pluralising-culture-alignment-for-large","slug":"self-pluralising-culture-alignment-for-large","title":"Self-Pluralising Culture Alignment for Large Language Models","date":"2024-10-16","arxiv_id":"2410.12971","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/self-pluralising-culture-alignment-for-large#ran","syntology_url":"https://syntology.ai/paper/2410.12971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12971"}},"official":{"repos":["shaoyangxu/culturespa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-world-simulator-crafting-physical","slug":"towards-world-simulator-crafting-physical","title":"Towards World Simulator: Crafting Physical Commonsense-Based Benchmark for Video Generation","date":"2024-10-07","arxiv_id":"2410.05363","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-world-simulator-crafting-physical#ran","syntology_url":"https://syntology.ai/paper/2410.05363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05363"}},"official":{"repos":["opengvlab/phygenbench"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-token-generation-in-large","slug":"counterfactual-token-generation-in-large","title":"Counterfactual Token Generation in Large Language Models","date":"2024-09-25","arxiv_id":"2409.17027","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-token-generation-in-large#ran","syntology_url":"https://syntology.ai/paper/2409.17027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17027"}},"official":{"repos":["networks-learning/counterfactual-llms"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/minstrel-structural-prompt-generation-with","slug":"minstrel-structural-prompt-generation-with","title":"Minstrel: Structural Prompt Generation with Multi-Agents Coordination for Non-AI Experts","date":"2024-09-20","arxiv_id":"2409.13449","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/minstrel-structural-prompt-generation-with#ran","syntology_url":"https://syntology.ai/paper/2409.13449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.13449"}},"official":{"repos":["sci-m-wang/minstrel"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-fully-autonomous-research-powered-by","slug":"towards-fully-autonomous-research-powered-by","title":"Toward Automated Simulation Research Workflow through LLM Prompt Engineering Design","date":"2024-08-28","arxiv_id":"2408.15512","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-fully-autonomous-research-powered-by#ran","syntology_url":"https://syntology.ai/paper/2408.15512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15512"}},"official":{"repos":["zokaraa/autonomous_simulation_agent"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/what-do-you-want-user-centric-prompt","slug":"what-do-you-want-user-centric-prompt","title":"What Do You Want? User-centric Prompt Generation for Text-to-image Synthesis via Multi-turn Guidance","date":"2024-08-23","arxiv_id":"2408.12910","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-do-you-want-user-centric-prompt#ran","syntology_url":"https://syntology.ai/paper/2408.12910","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12910"}},"official":{"repos":["superboom/dialprompt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-verilogeval-newer-llms-in-context","slug":"revisiting-verilogeval-newer-llms-in-context","title":"Revisiting VerilogEval: A Year of Improvements in Large-Language Models for Hardware Code Generation","date":"2024-08-20","arxiv_id":"2408.11053","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/revisiting-verilogeval-newer-llms-in-context#ran","syntology_url":"https://syntology.ai/paper/2408.11053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11053"}},"official":{"repos":["nvlabs/verilog-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/category-extensible-out-of-distribution-1","slug":"category-extensible-out-of-distribution-1","title":"Category-Extensible Out-of-Distribution Detection via Hierarchical Context Descriptions","date":"2024-07-23","arxiv_id":"2407.16725","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/category-extensible-out-of-distribution-1#ran","syntology_url":"https://syntology.ai/paper/2407.16725","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16725"}},"official":{"repos":["alibaba/catex"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lca-on-the-line-benchmarking-out-of","slug":"lca-on-the-line-benchmarking-out-of","title":"LCA-on-the-Line: Benchmarking Out-of-Distribution Generalization with Class Taxonomies","date":"2024-07-22","arxiv_id":"2407.16067","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lca-on-the-line-benchmarking-out-of#ran","syntology_url":"https://syntology.ai/paper/2407.16067","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16067"}},"official":{"repos":["elvishelvis/lca-on-the-line"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-design-and-analysis-of-llm-based","slug":"on-the-design-and-analysis-of-llm-based","title":"On the Design and Analysis of LLM-Based Algorithms","date":"2024-07-20","arxiv_id":"2407.14788","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-the-design-and-analysis-of-llm-based#ran","syntology_url":"https://syntology.ai/paper/2407.14788","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14788"}},"official":{"repos":["modelscope/agentscope"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tokenshap-interpreting-large-language-models","slug":"tokenshap-interpreting-large-language-models","title":"TokenSHAP: Interpreting Large Language Models with Monte Carlo Shapley Value Estimation","date":"2024-07-14","arxiv_id":"2407.10114","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/tokenshap-interpreting-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2407.10114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10114"}},"official":{"repos":["ronigold/TokenSHAP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/lapt-label-driven-automated-prompt-tuning-for","slug":"lapt-label-driven-automated-prompt-tuning-for","title":"LAPT: Label-driven Automated Prompt Tuning for OOD Detection with Vision-Language Models","date":"2024-07-12","arxiv_id":"2407.08966","repositories_listed":2,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":11,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":2,"n_no_contract":9,"n_pointer_only":4,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 2 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lapt-label-driven-automated-prompt-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2407.08966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08966"}},"official":{"repos":["ybzh/lapt"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/retrieval-augmented-generation-in","slug":"retrieval-augmented-generation-in","title":"Retrieval-augmented generation in multilingual settings","date":"2024-07-01","arxiv_id":"2407.01463","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/retrieval-augmented-generation-in#ran","syntology_url":"https://syntology.ai/paper/2407.01463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01463"}},"official":{"repos":["naver/bergen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-discrete-prompt-optimization-for-diffusion","slug":"on-discrete-prompt-optimization-for-diffusion","title":"On Discrete Prompt Optimization for Diffusion Models","date":"2024-06-27","arxiv_id":"2407.01606","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-discrete-prompt-optimization-for-diffusion#ran","syntology_url":"https://syntology.ai/paper/2407.01606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01606"}},"official":{"repos":["ruocwang/dpo-diffusion"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/self-reflection-outcome-is-sensitive-to","slug":"self-reflection-outcome-is-sensitive-to","title":"Self-Reflection Outcome is Sensitive to Prompt Construction","date":"2024-06-14","arxiv_id":"2406.10400","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-reflection-outcome-is-sensitive-to#ran","syntology_url":"https://syntology.ai/paper/2406.10400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10400"}},"official":{"repos":["michael98liu/mixture-of-prompts"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-scrutiny-detecting-backdoor-attacks","slug":"chain-of-scrutiny-detecting-backdoor-attacks","title":"Chain-of-Scrutiny: Detecting Backdoor Attacks for Large Language Models","date":"2024-06-10","arxiv_id":"2406.05948","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/chain-of-scrutiny-detecting-backdoor-attacks#ran","syntology_url":"https://syntology.ai/paper/2406.05948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05948"}},"official":{"repos":["lixi1994/CoS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/planning-like-human-a-dual-process-framework","slug":"planning-like-human-a-dual-process-framework","title":"Planning Like Human: A Dual-process Framework for Dialogue Planning","date":"2024-06-08","arxiv_id":"2406.05374","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/planning-like-human-a-dual-process-framework#ran","syntology_url":"https://syntology.ai/paper/2406.05374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05374"}},"official":{"repos":["cs-holder/DPDP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pace-parsimonious-concept-engineering-for","slug":"pace-parsimonious-concept-engineering-for","title":"PaCE: Parsimonious Concept Engineering for Large Language Models","date":"2024-06-06","arxiv_id":"2406.04331","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pace-parsimonious-concept-engineering-for#ran","syntology_url":"https://syntology.ai/paper/2406.04331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04331"}},"official":{"repos":["peterljq/parsimonious-concept-engineering"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-prompt-report-a-systematic-survey-of","slug":"the-prompt-report-a-systematic-survey-of","title":"The Prompt Report: A Systematic Survey of Prompting Techniques","date":"2024-06-06","arxiv_id":"2406.06608","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-prompt-report-a-systematic-survey-of#ran","syntology_url":"https://syntology.ai/paper/2406.06608","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06608"}},"official":{"repos":["trigaten/Prompt_Systematic_Review","trigaten/the_prompt_report","trigaten/learn_prompting"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/interpretabnet-distilling-predictive-signals","slug":"interpretabnet-distilling-predictive-signals","title":"InterpreTabNet: Distilling Predictive Signals from Tabular Data by Salient Feature Interpretation","date":"2024-06-01","arxiv_id":"2406.00426","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":10,"n_ran_checked":10,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 10 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified; every one of the 10 samples that ran constructed an object rather than computing a result","sample_list":"/paper/interpretabnet-distilling-predictive-signals#ran","syntology_url":"https://syntology.ai/paper/2406.00426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00426"}},"official":{"repos":["jacobyhsi/InterpreTabNet"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":10,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/easy-problems-that-llms-get-wrong","slug":"easy-problems-that-llms-get-wrong","title":"Easy Problems That LLMs Get Wrong","date":"2024-05-30","arxiv_id":"2405.19616","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/easy-problems-that-llms-get-wrong#ran","syntology_url":"https://syntology.ai/paper/2405.19616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19616"}},"official":{"repos":["autogenai/easy-problems-that-llms-get-wrong"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-graph-learning-improve-task-planning","slug":"can-graph-learning-improve-task-planning","title":"Can Graph Learning Improve Planning in LLM-based Agents?","date":"2024-05-29","arxiv_id":"2405.19119","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-graph-learning-improve-task-planning#ran","syntology_url":"https://syntology.ai/paper/2405.19119","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19119"}},"official":{"repos":["wxxshirley/gnn4taskplan"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-in-conversation-team-building-for","slug":"adaptive-in-conversation-team-building-for","title":"Adaptive In-conversation Team Building for Language Model Agents","date":"2024-05-29","arxiv_id":"2405.19425","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/adaptive-in-conversation-team-building-for#ran","syntology_url":"https://syntology.ai/paper/2405.19425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19425"}},"official":{"repos":["ag2ai/ag2"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/orlm-training-large-language-models-for","slug":"orlm-training-large-language-models-for","title":"ORLM: A Customizable Framework in Training Large Models for Automated Optimization Modeling","date":"2024-05-28","arxiv_id":"2405.17743","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/orlm-training-large-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2405.17743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17743"}},"official":{"repos":["cardinal-operations/orlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/promptwizard-task-aware-agent-driven-prompt","slug":"promptwizard-task-aware-agent-driven-prompt","title":"PromptWizard: Task-Aware Prompt Optimization Framework","date":"2024-05-28","arxiv_id":"2405.18369","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/promptwizard-task-aware-agent-driven-prompt#ran","syntology_url":"https://syntology.ai/paper/2405.18369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18369"}},"official":null}},{"url":"/paper/depth-prompting-for-sensor-agnostic-depth","slug":"depth-prompting-for-sensor-agnostic-depth","title":"Depth Prompting for Sensor-Agnostic Depth Estimation","date":"2024-05-20","arxiv_id":"2405.11867","repositories_listed":0,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/depth-prompting-for-sensor-agnostic-depth#ran","syntology_url":"https://syntology.ai/paper/2405.11867","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11867"}},"official":null}},{"url":"/paper/cityllava-efficient-fine-tuning-for-vlms-in","slug":"cityllava-efficient-fine-tuning-for-vlms-in","title":"CityLLaVA: Efficient Fine-Tuning for VLMs in City Scenario","date":"2024-05-06","arxiv_id":"2405.03194","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cityllava-efficient-fine-tuning-for-vlms-in#ran","syntology_url":"https://syntology.ai/paper/2405.03194","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.03194"}},"official":{"repos":["alibaba/aicity2024_track2_aliopentrek_cityllava"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/probabilistic-inference-in-language-models","slug":"probabilistic-inference-in-language-models","title":"Probabilistic Inference in Language Models via Twisted Sequential Monte Carlo","date":"2024-04-26","arxiv_id":"2404.17546","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":17,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/probabilistic-inference-in-language-models#ran","syntology_url":"https://syntology.ai/paper/2404.17546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.17546"}},"official":{"repos":["silent-zebra/twisted-smc-lm"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/semantic-routing-for-enhanced-performance-of","slug":"semantic-routing-for-enhanced-performance-of","title":"Semantic Routing for Enhanced Performance of LLM-Assisted Intent-Based 5G Core Network Management and Orchestration","date":"2024-04-24","arxiv_id":"2404.15869","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/semantic-routing-for-enhanced-performance-of#ran","syntology_url":"https://syntology.ai/paper/2404.15869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15869"}},"official":null}},{"url":"/paper/test-time-adaptation-with-salip-a-cascade-of","slug":"test-time-adaptation-with-salip-a-cascade-of","title":"Test-Time Adaptation with SaLIP: A Cascade of SAM and CLIP for Zero shot Medical Image Segmentation","date":"2024-04-09","arxiv_id":"2404.06362","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/test-time-adaptation-with-salip-a-cascade-of#ran","syntology_url":"https://syntology.ai/paper/2404.06362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.06362"}},"official":{"repos":["aleemsidra/SaLIP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/promptad-learning-prompts-with-only-normal","slug":"promptad-learning-prompts-with-only-normal","title":"PromptAD: Learning Prompts with only Normal Samples for Few-Shot Anomaly Detection","date":"2024-04-08","arxiv_id":"2404.05231","repositories_listed":1,"syntology":{"n":16,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":10,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/promptad-learning-prompts-with-only-normal#ran","syntology_url":"https://syntology.ai/paper/2404.05231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05231"}},"official":{"repos":["funz-0/promptad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/utebc-nlp-at-semeval-2024-task-9-can-llms-be","slug":"utebc-nlp-at-semeval-2024-task-9-can-llms-be","title":"uTeBC-NLP at SemEval-2024 Task 9: Can LLMs be Lateral Thinkers?","date":"2024-04-03","arxiv_id":"2404.02474","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/utebc-nlp-at-semeval-2024-task-9-can-llms-be#ran","syntology_url":"https://syntology.ai/paper/2404.02474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02474"}},"official":{"repos":["ipouyall/can-llms-be-lateral-thinkers"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompt-learning-via-meta-regularization","slug":"prompt-learning-via-meta-regularization","title":"Prompt Learning via Meta-Regularization","date":"2024-04-01","arxiv_id":"2404.00851","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prompt-learning-via-meta-regularization#ran","syntology_url":"https://syntology.ai/paper/2404.00851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00851"}},"official":{"repos":["mlvlab/prometar"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/just-shift-it-test-time-prototype-shifting","slug":"just-shift-it-test-time-prototype-shifting","title":"Just Shift It: Test-Time Prototype Shifting for Zero-Shot Generalization with Vision-Language Models","date":"2024-03-19","arxiv_id":"2403.12952","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/just-shift-it-test-time-prototype-shifting#ran","syntology_url":"https://syntology.ai/paper/2403.12952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12952"}},"official":{"repos":["elaine-sui/tps"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-powered-context-aware","slug":"large-language-models-powered-context-aware","title":"Large Language Models Powered Context-aware Motion Prediction in Autonomous Driving","date":"2024-03-17","arxiv_id":"2403.11057","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-powered-context-aware#ran","syntology_url":"https://syntology.ai/paper/2403.11057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11057"}},"official":{"repos":["air-discover/llm-augmented-mtr"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-ecg-classification-with-multimodal","slug":"zero-shot-ecg-classification-with-multimodal","title":"Zero-Shot ECG Classification with Multimodal Learning and Test-time Clinical Knowledge Enhancement","date":"2024-03-11","arxiv_id":"2403.06659","repositories_listed":2,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/zero-shot-ecg-classification-with-multimodal#ran","syntology_url":"https://syntology.ai/paper/2403.06659","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.06659"}},"official":{"repos":["cheliu-computation/merl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/can-llms-separate-instructions-from-data-and","slug":"can-llms-separate-instructions-from-data-and","title":"Can LLMs Separate Instructions From Data? And What Do We Even Mean By That?","date":"2024-03-11","arxiv_id":"2403.06833","repositories_listed":2,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-llms-separate-instructions-from-data-and#ran","syntology_url":"https://syntology.ai/paper/2403.06833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.06833"}},"official":{"repos":["egozverev/shold-it-be-executed-or-processed"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vidprom-a-million-scale-real-prompt-gallery","slug":"vidprom-a-million-scale-real-prompt-gallery","title":"VidProM: A Million-scale Real Prompt-Gallery Dataset for Text-to-Video Diffusion Models","date":"2024-03-10","arxiv_id":"2403.06098","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vidprom-a-million-scale-real-prompt-gallery#ran","syntology_url":"https://syntology.ai/paper/2403.06098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.06098"}},"official":{"repos":["wangwenhao0716/vidprom"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/erbench-an-entity-relationship-based","slug":"erbench-an-entity-relationship-based","title":"ERBench: An Entity-Relationship based Automatically Verifiable Hallucination Benchmark for Large Language Models","date":"2024-03-08","arxiv_id":"2403.05266","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/erbench-an-entity-relationship-based#ran","syntology_url":"https://syntology.ai/paper/2403.05266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05266"}},"official":{"repos":["dilab-kaist/erbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/promoai-process-modeling-with-generative-ai","slug":"promoai-process-modeling-with-generative-ai","title":"ProMoAI: Process Modeling with Generative AI","date":"2024-03-07","arxiv_id":"2403.04327","repositories_listed":0,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/promoai-process-modeling-with-generative-ai#ran","syntology_url":"https://syntology.ai/paper/2403.04327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04327"}},"official":null}},{"url":"/paper/cogbench-a-large-language-model-walks-into-a","slug":"cogbench-a-large-language-model-walks-into-a","title":"CogBench: a large language model walks into a psychology lab","date":"2024-02-28","arxiv_id":"2402.18225","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cogbench-a-large-language-model-walks-into-a#ran","syntology_url":"https://syntology.ai/paper/2402.18225","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18225"}},"official":{"repos":["juliancodaforno/cogbench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-pro-learning-to-evolve-via-policy-level","slug":"agent-pro-learning-to-evolve-via-policy-level","title":"Agent-Pro: Learning to Evolve via Policy-Level Reflection and Optimization","date":"2024-02-27","arxiv_id":"2402.17574","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agent-pro-learning-to-evolve-via-policy-level#ran","syntology_url":"https://syntology.ai/paper/2402.17574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17574"}},"official":{"repos":["zwq2018/agent-pro"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/language-agents-as-optimizable-graphs","slug":"language-agents-as-optimizable-graphs","title":"Language Agents as Optimizable Graphs","date":"2024-02-26","arxiv_id":"2402.16823","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-agents-as-optimizable-graphs#ran","syntology_url":"https://syntology.ai/paper/2402.16823","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16823"}},"official":{"repos":["metauto-ai/gptswarm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/how-interpretable-are-reasoning-explanations","slug":"how-interpretable-are-reasoning-explanations","title":"How Interpretable are Reasoning Explanations from Prompting Large Language Models?","date":"2024-02-19","arxiv_id":"2402.11863","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":16,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/how-interpretable-are-reasoning-explanations#ran","syntology_url":"https://syntology.ai/paper/2402.11863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11863"}},"official":{"repos":["wj210/cot_interpretability"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/self-augmented-in-context-learning-for","slug":"self-augmented-in-context-learning-for","title":"Self-Augmented In-Context Learning for Unsupervised Word Translation","date":"2024-02-15","arxiv_id":"2402.10024","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/self-augmented-in-context-learning-for#ran","syntology_url":"https://syntology.ai/paper/2402.10024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10024"}},"official":{"repos":["cambridgeltl/sail-bli"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/culturellm-incorporating-cultural-differences","slug":"culturellm-incorporating-cultural-differences","title":"CultureLLM: Incorporating Cultural Differences into Large Language Models","date":"2024-02-09","arxiv_id":"2402.10946","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/culturellm-incorporating-cultural-differences#ran","syntology_url":"https://syntology.ai/paper/2402.10946","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10946"}},"official":{"repos":["scarelette/culturellm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-effect-of-sampling-temperature-on-problem","slug":"the-effect-of-sampling-temperature-on-problem","title":"The Effect of Sampling Temperature on Problem Solving in Large Language Models","date":"2024-02-07","arxiv_id":"2402.05201","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-effect-of-sampling-temperature-on-problem#ran","syntology_url":"https://syntology.ai/paper/2402.05201","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05201"}},"official":{"repos":["matthewrenze/jhu-llm-temperature"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/intent-based-prompt-calibration-enhancing","slug":"intent-based-prompt-calibration-enhancing","title":"Intent-based Prompt Calibration: Enhancing prompt optimization with synthetic boundary cases","date":"2024-02-05","arxiv_id":"2402.03099","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intent-based-prompt-calibration-enhancing#ran","syntology_url":"https://syntology.ai/paper/2402.03099","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03099"}},"official":{"repos":["eladlev/autoprompt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhance-reasoning-for-large-language-models","slug":"enhance-reasoning-for-large-language-models","title":"Enhance Reasoning for Large Language Models in the Game Werewolf","date":"2024-02-04","arxiv_id":"2402.02330","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhance-reasoning-for-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2402.02330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02330"}},"official":{"repos":["boluoweifenda/werewolf"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/style-vectors-for-steering-generative-large","slug":"style-vectors-for-steering-generative-large","title":"Style Vectors for Steering Generative Large Language Model","date":"2024-02-02","arxiv_id":"2402.01618","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/style-vectors-for-steering-generative-large#ran","syntology_url":"https://syntology.ai/paper/2402.01618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01618"}},"official":{"repos":["dlr-sc/style-vectors-for-steering-llms"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/llms-learn-governing-principles-of-dynamical","slug":"llms-learn-governing-principles-of-dynamical","title":"LLMs learn governing principles of dynamical systems, revealing an in-context neural scaling law","date":"2024-02-01","arxiv_id":"2402.00795","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llms-learn-governing-principles-of-dynamical#ran","syntology_url":"https://syntology.ai/paper/2402.00795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00795"}},"official":{"repos":["AntonioLiu97/llmICL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/question-answer-cross-language-image-matching","slug":"question-answer-cross-language-image-matching","title":"Question-Answer Cross Language Image Matching for Weakly Supervised Semantic Segmentation","date":"2024-01-18","arxiv_id":"2401.09883","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/question-answer-cross-language-image-matching#ran","syntology_url":"https://syntology.ai/paper/2401.09883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09883"}},"official":{"repos":["cvi-szu/qa-clims"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/code-generation-with-alphacodium-from-prompt","slug":"code-generation-with-alphacodium-from-prompt","title":"Code Generation with AlphaCodium: From Prompt Engineering to Flow Engineering","date":"2024-01-16","arxiv_id":"2401.08500","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/code-generation-with-alphacodium-from-prompt#ran","syntology_url":"https://syntology.ai/paper/2401.08500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08500"}},"official":{"repos":["codium-ai/alphacodium"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-benefits-of-a-concise-chain-of-thought-on","slug":"the-benefits-of-a-concise-chain-of-thought-on","title":"The Benefits of a Concise Chain of Thought on Problem-Solving in Large Language Models","date":"2024-01-11","arxiv_id":"2401.05618","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-benefits-of-a-concise-chain-of-thought-on#ran","syntology_url":"https://syntology.ai/paper/2401.05618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05618"}},"official":{"repos":["matthewrenze/jhu-concise-cot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/make-prompts-adaptable-bayesian-modeling-for","slug":"make-prompts-adaptable-bayesian-modeling-for","title":"Make Prompts Adaptable: Bayesian Modeling for Vision-Language Prompt Learning with Data-Dependent Prior","date":"2024-01-09","arxiv_id":"2401.06799","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/make-prompts-adaptable-bayesian-modeling-for#ran","syntology_url":"https://syntology.ai/paper/2401.06799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06799"}},"official":{"repos":["youngjae-cho/app"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-prompt-with-text-only-supervision","slug":"learning-to-prompt-with-text-only-supervision","title":"Learning to Prompt with Text Only Supervision for Vision-Language Models","date":"2024-01-04","arxiv_id":"2401.02418","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-prompt-with-text-only-supervision#ran","syntology_url":"https://syntology.ai/paper/2401.02418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02418"}},"official":{"repos":["muzairkhattak/protext"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/supervised-knowledge-makes-large-language","slug":"supervised-knowledge-makes-large-language","title":"Supervised Knowledge Makes Large Language Models Better In-context Learners","date":"2023-12-26","arxiv_id":"2312.15918","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/supervised-knowledge-makes-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.15918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15918"}},"official":{"repos":["yanglinyi/supervised-knowledge-makes-large-language-models-better-in-context-learners"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/agentcoder-multi-agent-based-code-generation","slug":"agentcoder-multi-agent-based-code-generation","title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","date":"2023-12-20","arxiv_id":"2312.13010","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agentcoder-multi-agent-based-code-generation#ran","syntology_url":"https://syntology.ai/paper/2312.13010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13010"}},"official":{"repos":["huangd1999/AgentCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompt-based-distribution-alignment-for","slug":"prompt-based-distribution-alignment-for","title":"Prompt-based Distribution Alignment for Unsupervised Domain Adaptation","date":"2023-12-15","arxiv_id":"2312.09553","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prompt-based-distribution-alignment-for#ran","syntology_url":"https://syntology.ai/paper/2312.09553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09553"}},"official":{"repos":["baishuanghao/prompt-based-distribution-alignment"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mothernet-a-foundational-hypernetwork-for","slug":"mothernet-a-foundational-hypernetwork-for","title":"MotherNet: Fast Training and Inference via Hyper-Network Transformers","date":"2023-12-14","arxiv_id":"2312.08598","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mothernet-a-foundational-hypernetwork-for#ran","syntology_url":"https://syntology.ai/paper/2312.08598","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08598"}},"official":{"repos":["microsoft/ticl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-hierarchical-prompt-with-structured","slug":"learning-hierarchical-prompt-with-structured","title":"Learning Hierarchical Prompt with Structured Linguistic Knowledge for Vision-Language Models","date":"2023-12-11","arxiv_id":"2312.06323","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-hierarchical-prompt-with-structured#ran","syntology_url":"https://syntology.ai/paper/2312.06323","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06323"}},"official":{"repos":["vill-lab/2024-aaai-hpt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/can-generalist-foundation-models-outcompete","slug":"can-generalist-foundation-models-outcompete","title":"Can Generalist Foundation Models Outcompete Special-Purpose Tuning? Case Study in Medicine","date":"2023-11-28","arxiv_id":"2311.16452","repositories_listed":2,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/can-generalist-foundation-models-outcompete#ran","syntology_url":"https://syntology.ai/paper/2311.16452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16452"}},"official":null}},{"url":"/paper/beyond-sole-strength-customized-ensembles-for","slug":"beyond-sole-strength-customized-ensembles-for","title":"Beyond Sole Strength: Customized Ensembles for Generalized Vision-Language Models","date":"2023-11-28","arxiv_id":"2311.17091","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/beyond-sole-strength-customized-ensembles-for#ran","syntology_url":"https://syntology.ai/paper/2311.17091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17091"}},"official":{"repos":["zhihelu/ensemble_vlm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dp-opt-make-large-language-model-your-privacy","slug":"dp-opt-make-large-language-model-your-privacy","title":"DP-OPT: Make Large Language Model Your Privacy-Preserving Prompt Engineer","date":"2023-11-27","arxiv_id":"2312.03724","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dp-opt-make-large-language-model-your-privacy#ran","syntology_url":"https://syntology.ai/paper/2312.03724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03724"}},"official":{"repos":["vita-group/dp-opt"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/simulating-opinion-dynamics-with-networks-of","slug":"simulating-opinion-dynamics-with-networks-of","title":"Simulating Opinion Dynamics with Networks of LLM-based Agents","date":"2023-11-16","arxiv_id":"2311.09618","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simulating-opinion-dynamics-with-networks-of#ran","syntology_url":"https://syntology.ai/paper/2311.09618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09618"}},"official":{"repos":["yunshiuan/llm-agent-opinion-dynamics"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-are-prompts-different-in-terms-of","slug":"how-are-prompts-different-in-terms-of","title":"How are Prompts Different in Terms of Sensitivity?","date":"2023-11-13","arxiv_id":"2311.07230","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-are-prompts-different-in-terms-of#ran","syntology_url":"https://syntology.ai/paper/2311.07230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07230"}},"official":{"repos":["ukplab/naacl2024-prompt-sensitivity"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/assessing-logical-puzzle-solving-in-large","slug":"assessing-logical-puzzle-solving-in-large","title":"Assessing Logical Puzzle Solving in Large Language Models: Insights from a Minesweeper Case Study","date":"2023-11-13","arxiv_id":"2311.07387","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/assessing-logical-puzzle-solving-in-large#ran","syntology_url":"https://syntology.ai/paper/2311.07387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07387"}},"official":{"repos":["yinghao-li/minesweeper-for-llm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-distillation-makes-large-language","slug":"instruction-distillation-makes-large-language","title":"Instruction Distillation Makes Large Language Models Efficient Zero-shot Rankers","date":"2023-11-02","arxiv_id":"2311.01555","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/instruction-distillation-makes-large-language#ran","syntology_url":"https://syntology.ai/paper/2311.01555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01555"}},"official":{"repos":["sunnweiwei/rankgpt"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-for-aspect-based","slug":"large-language-models-for-aspect-based","title":"Large language models for aspect-based sentiment analysis","date":"2023-10-27","arxiv_id":"2310.18025","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-for-aspect-based#ran","syntology_url":"https://syntology.ai/paper/2310.18025","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18025"}},"official":{"repos":["qagentur/absa_llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cxr-llava-multimodal-large-language-model-for","slug":"cxr-llava-multimodal-large-language-model-for","title":"CXR-LLAVA: a multimodal large language model for interpreting chest X-ray images","date":"2023-10-22","arxiv_id":"2310.18341","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cxr-llava-multimodal-large-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2310.18341","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18341"}},"official":{"repos":["ecofri/cxr_llava"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-bilingual-lexicon-induction-with-large","slug":"on-bilingual-lexicon-induction-with-large","title":"On Bilingual Lexicon Induction with Large Language Models","date":"2023-10-21","arxiv_id":"2310.13995","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-bilingual-lexicon-induction-with-large#ran","syntology_url":"https://syntology.ai/paper/2310.13995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13995"}},"official":{"repos":["cambridgeltl/prompt4bli"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-models-are-zero-shot-reward","slug":"vision-language-models-are-zero-shot-reward","title":"Vision-Language Models are Zero-Shot Reward Models for Reinforcement Learning","date":"2023-10-19","arxiv_id":"2310.12921","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vision-language-models-are-zero-shot-reward#ran","syntology_url":"https://syntology.ai/paper/2310.12921","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12921"}},"official":{"repos":["alignmentresearch/vlmrm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-model-empowered-agents-for","slug":"large-language-model-empowered-agents-for","title":"EconAgent: Large Language Model-Empowered Agents for Simulating Macroeconomic Activities","date":"2023-10-16","arxiv_id":"2310.10436","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/large-language-model-empowered-agents-for#ran","syntology_url":"https://syntology.ai/paper/2310.10436","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10436"}},"official":{"repos":["tsinghua-fib-lab/acl24-econagent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-factuality-a-comprehensive-evaluation","slug":"beyond-factuality-a-comprehensive-evaluation","title":"Beyond Factuality: A Comprehensive Evaluation of Large Language Models as Knowledge Generators","date":"2023-10-11","arxiv_id":"2310.07289","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":14,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/beyond-factuality-a-comprehensive-evaluation#ran","syntology_url":"https://syntology.ai/paper/2310.07289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07289"}},"official":{"repos":["chanliang/conner"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-natural-language-inference-for","slug":"chain-of-natural-language-inference-for","title":"Chain of Natural Language Inference for Reducing Large Language Model Ungrounded Hallucinations","date":"2023-10-06","arxiv_id":"2310.03951","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chain-of-natural-language-inference-for#ran","syntology_url":"https://syntology.ai/paper/2310.03951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03951"}},"official":{"repos":["microsoft/conli_hallucination"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/thought-propagation-an-analogical-approach-to","slug":"thought-propagation-an-analogical-approach-to","title":"Thought Propagation: An Analogical Approach to Complex Reasoning with Large Language Models","date":"2023-10-06","arxiv_id":"2310.03965","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/thought-propagation-an-analogical-approach-to#ran","syntology_url":"https://syntology.ai/paper/2310.03965","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03965"}},"official":{"repos":["Samyu0304/thought-propagation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-s-the-magic-word-a-control-theory-of-llm","slug":"what-s-the-magic-word-a-control-theory-of-llm","title":"What's the Magic Word? A Control Theory of LLM Prompting","date":"2023-10-02","arxiv_id":"2310.04444","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-s-the-magic-word-a-control-theory-of-llm#ran","syntology_url":"https://syntology.ai/paper/2310.04444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04444"}},"official":{"repos":["amanb2000/magic_words"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rladapter-bridging-large-language-models-to","slug":"rladapter-bridging-large-language-models-to","title":"AdaRefiner: Refining Decisions of Language Models with Adaptive Feedback","date":"2023-09-29","arxiv_id":"2309.17176","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rladapter-bridging-large-language-models-to#ran","syntology_url":"https://syntology.ai/paper/2309.17176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17176"}},"official":{"repos":["pku-rl/adarefiner"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/suspicion-agent-playing-imperfect-information","slug":"suspicion-agent-playing-imperfect-information","title":"Suspicion-Agent: Playing Imperfect Information Games with Theory of Mind Aware GPT-4","date":"2023-09-29","arxiv_id":"2309.17277","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/suspicion-agent-playing-imperfect-information#ran","syntology_url":"https://syntology.ai/paper/2309.17277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17277"}},"official":{"repos":["cr-gjx/suspicion-agent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dept-decoupled-prompt-tuning","slug":"dept-decoupled-prompt-tuning","title":"DePT: Decoupled Prompt Tuning","date":"2023-09-14","arxiv_id":"2309.07439","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/dept-decoupled-prompt-tuning#ran","syntology_url":"https://syntology.ai/paper/2309.07439","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07439"}},"official":{"repos":["koorye/dept"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-prompt-evaluation-and-optimization","slug":"offline-prompt-evaluation-and-optimization","title":"Query-Dependent Prompt Evaluation and Optimization with Offline Inverse RL","date":"2023-09-13","arxiv_id":"2309.06553","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/offline-prompt-evaluation-and-optimization#ran","syntology_url":"https://syntology.ai/paper/2309.06553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06553"}},"official":{"repos":["holarissun/prompt-oirl","vanderschaarlab/prompt-oirl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/read-only-prompt-optimization-for-vision","slug":"read-only-prompt-optimization-for-vision","title":"Read-only Prompt Optimization for Vision-Language Few-shot Learning","date":"2023-08-29","arxiv_id":"2308.14960","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/read-only-prompt-optimization-for-vision#ran","syntology_url":"https://syntology.ai/paper/2308.14960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14960"}},"official":{"repos":["mlvlab/rpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"1b44512a23521ea2c59c0513a5e8d687e73d05be572fbdb4166a27f107d4e412","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}