{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/prompt-engineering/papers/3","list_of":"/task/prompt-engineering","task":"Prompt Engineering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":13,"rows_per_page":100,"rows":[201,300],"of":1236,"counts":{"archive_papers_tagged":1236,"with_a_code_link":454,"where_syntology_ran_a_sample":143,"not_listed_spam_title":0,"listed":1236,"listed_where_code_ran":143,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":118,"every_run_a_failure_of_syntologys_instrument":25,"listed_with_a_run_with_no_instrument_failure":118,"listed_every_run_a_failure_of_syntologys_instrument":25,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/prompt-engineering","prev":"/task/prompt-engineering/papers/2","next":"/task/prompt-engineering/papers/4","papers":[{"url":"/paper/fine-tuning-large-language-models-for-entity","slug":"fine-tuning-large-language-models-for-entity","title":"Fine-tuning Large Language Models for Entity Matching","date":"2024-09-12","arxiv_id":"2409.08185","repositories_listed":1,"syntology":null},{"url":"/paper/llm-honeypot-leveraging-large-language-models","slug":"llm-honeypot-leveraging-large-language-models","title":"LLM Honeypot: Leveraging Large Language Models as Advanced Interactive Honeypot Systems","date":"2024-09-12","arxiv_id":"2409.08234","repositories_listed":1,"syntology":null},{"url":"/paper/drllm-prompt-enhanced-distributed-denial-of","slug":"drllm-prompt-enhanced-distributed-denial-of","title":"DrLLM: Prompt-Enhanced Distributed Denial-of-Service Resistance Method with Large Language Models","date":"2024-09-11","arxiv_id":"2409.10561","repositories_listed":1,"syntology":null},{"url":"/paper/insights-from-benchmarking-frontier-language","slug":"insights-from-benchmarking-frontier-language","title":"Insights from Benchmarking Frontier Language Models on Web App Code Generation","date":"2024-09-08","arxiv_id":"2409.05177","repositories_listed":1,"syntology":null},{"url":"/paper/2409-13704","slug":"2409-13704","title":"Entity Extraction from High-Level Corruption Schemes via Large Language Models","date":"2024-09-05","arxiv_id":"2409.13704","repositories_listed":1,"syntology":null},{"url":"/paper/bootstrap-segmentation-foundation-model-under","slug":"bootstrap-segmentation-foundation-model-under","title":"Bootstrap Segmentation Foundation Model under Distribution Shift via Object-Centric Learning","date":"2024-08-29","arxiv_id":"2408.16310","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-named-entity-recognition-using-few","slug":"evaluating-named-entity-recognition-using-few","title":"Evaluating Named Entity Recognition Using Few-Shot Prompting with Large Language Models","date":"2024-08-28","arxiv_id":"2408.15796","repositories_listed":1,"syntology":null},{"url":"/paper/towards-fully-autonomous-research-powered-by","slug":"towards-fully-autonomous-research-powered-by","title":"Toward Automated Simulation Research Workflow through LLM Prompt Engineering Design","date":"2024-08-28","arxiv_id":"2408.15512","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-fully-autonomous-research-powered-by#ran","syntology_url":"https://syntology.ai/paper/2408.15512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15512"}},"official":{"repos":["zokaraa/autonomous_simulation_agent"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-and-large-language-model","slug":"vision-language-and-large-language-model","title":"Vision-Language and Large Language Model Performance in Gastroenterology: GPT, Claude, Llama, Phi, Mistral, Gemma, and Quantized Models","date":"2024-08-25","arxiv_id":"2409.00084","repositories_listed":1,"syntology":null},{"url":"/paper/what-do-you-want-user-centric-prompt","slug":"what-do-you-want-user-centric-prompt","title":"What Do You Want? User-centric Prompt Generation for Text-to-image Synthesis via Multi-turn Guidance","date":"2024-08-23","arxiv_id":"2408.12910","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-do-you-want-user-centric-prompt#ran","syntology_url":"https://syntology.ai/paper/2408.12910","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12910"}},"official":{"repos":["superboom/dialprompt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/controllable-text-generation-for-large","slug":"controllable-text-generation-for-large","title":"Controllable Text Generation for Large Language Models: A Survey","date":"2024-08-22","arxiv_id":"2408.12599","repositories_listed":1,"syntology":null},{"url":"/paper/epic-cost-effective-search-based-prompt","slug":"epic-cost-effective-search-based-prompt","title":"EPiC: Cost-effective Search-based Prompt Engineering of LLMs for Code Generation","date":"2024-08-20","arxiv_id":"2408.11198","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-verilogeval-newer-llms-in-context","slug":"revisiting-verilogeval-newer-llms-in-context","title":"Revisiting VerilogEval: A Year of Improvements in Large-Language Models for Hardware Code Generation","date":"2024-08-20","arxiv_id":"2408.11053","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/revisiting-verilogeval-newer-llms-in-context#ran","syntology_url":"https://syntology.ai/paper/2408.11053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11053"}},"official":{"repos":["nvlabs/verilog-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/v-roast-a-new-dataset-for-visual-road","slug":"v-roast-a-new-dataset-for-visual-road","title":"V-RoAst: Visual Road Assessment. Can VLM be a Road Safety Assessor Using the iRAP Standard?","date":"2024-08-20","arxiv_id":"2408.10872","repositories_listed":1,"syntology":null},{"url":"/paper/2408-02964","slug":"2408-02964","title":"Accuracy and Consistency of LLMs in the Registered Dietitian Exam: The Impact of Prompt Engineering and Knowledge Retrieval","date":"2024-08-06","arxiv_id":"2408.02964","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00727","slug":"2408-00727","title":"Improving Retrieval-Augmented Generation in Medicine with Iterative Follow-up Questions","date":"2024-08-01","arxiv_id":"2408.00727","repositories_listed":1,"syntology":null},{"url":"/paper/versusdebias-universal-zero-shot-debiasing","slug":"versusdebias-universal-zero-shot-debiasing","title":"VersusDebias: Universal Zero-Shot Debiasing for Text-to-Image Models via SLM-Based Prompt Engineering and Generative Adversary","date":"2024-07-28","arxiv_id":"2407.19524","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/versusdebias-universal-zero-shot-debiasing#ran","syntology_url":"https://syntology.ai/paper/2407.19524","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.19524"}},"official":{"repos":["versusdebias/versusdebias"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/segmentation-by-registration-enabled-sam","slug":"segmentation-by-registration-enabled-sam","title":"Segmentation by registration-enabled SAM prompt engineering using five reference images","date":"2024-07-25","arxiv_id":"2407.17933","repositories_listed":1,"syntology":null},{"url":"/paper/category-extensible-out-of-distribution-1","slug":"category-extensible-out-of-distribution-1","title":"Category-Extensible Out-of-Distribution Detection via Hierarchical Context Descriptions","date":"2024-07-23","arxiv_id":"2407.16725","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/category-extensible-out-of-distribution-1#ran","syntology_url":"https://syntology.ai/paper/2407.16725","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16725"}},"official":{"repos":["alibaba/catex"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/figure-it-out-analyzing-based-jailbreak","slug":"figure-it-out-analyzing-based-jailbreak","title":"LLMs can be Dangerous Reasoners: Analyzing-based Jailbreak Attack on Large Language Models","date":"2024-07-23","arxiv_id":"2407.16205","repositories_listed":1,"syntology":null},{"url":"/paper/lca-on-the-line-benchmarking-out-of","slug":"lca-on-the-line-benchmarking-out-of","title":"LCA-on-the-Line: Benchmarking Out-of-Distribution Generalization with Class Taxonomies","date":"2024-07-22","arxiv_id":"2407.16067","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lca-on-the-line-benchmarking-out-of#ran","syntology_url":"https://syntology.ai/paper/2407.16067","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16067"}},"official":{"repos":["elvishelvis/lca-on-the-line"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-design-and-analysis-of-llm-based","slug":"on-the-design-and-analysis-of-llm-based","title":"On the Design and Analysis of LLM-Based Algorithms","date":"2024-07-20","arxiv_id":"2407.14788","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-the-design-and-analysis-of-llm-based#ran","syntology_url":"https://syntology.ai/paper/2407.14788","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14788"}},"official":{"repos":["modelscope/agentscope"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/persllm-a-personified-training-approach-for","slug":"persllm-a-personified-training-approach-for","title":"PersLLM: A Personified Training Approach for Large Language Models","date":"2024-07-17","arxiv_id":"2407.12393","repositories_listed":1,"syntology":null},{"url":"/paper/tokenshap-interpreting-large-language-models","slug":"tokenshap-interpreting-large-language-models","title":"TokenSHAP: Interpreting Large Language Models with Monte Carlo Shapley Value Estimation","date":"2024-07-14","arxiv_id":"2407.10114","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/tokenshap-interpreting-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2407.10114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10114"}},"official":{"repos":["ronigold/TokenSHAP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/are-large-language-models-really-bias-free","slug":"are-large-language-models-really-bias-free","title":"Are Large Language Models Really Bias-Free? Jailbreak Prompts for Assessing Adversarial Robustness to Bias Elicitation","date":"2024-07-11","arxiv_id":"2407.08441","repositories_listed":1,"syntology":null},{"url":"/paper/virtual-agents-for-alcohol-use-counseling","slug":"virtual-agents-for-alcohol-use-counseling","title":"Virtual Agents for Alcohol Use Counseling: Exploring LLM-Powered Motivational Interviewing","date":"2024-07-10","arxiv_id":"2407.08095","repositories_listed":1,"syntology":null},{"url":"/paper/using-llms-to-label-medical-papers-according","slug":"using-llms-to-label-medical-papers-according","title":"Using LLMs to label medical papers according to the CIViC evidence model","date":"2024-07-05","arxiv_id":"2407.04466","repositories_listed":1,"syntology":null},{"url":"/paper/logeval-a-comprehensive-benchmark-suite-for","slug":"logeval-a-comprehensive-benchmark-suite-for","title":"LogEval: A Comprehensive Benchmark Suite for Large Language Models In Log Analysis","date":"2024-07-02","arxiv_id":"2407.01896","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-augmented-generation-in","slug":"retrieval-augmented-generation-in","title":"Retrieval-augmented generation in multilingual settings","date":"2024-07-01","arxiv_id":"2407.01463","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/retrieval-augmented-generation-in#ran","syntology_url":"https://syntology.ai/paper/2407.01463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01463"}},"official":{"repos":["naver/bergen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-for-power-scheduling-a","slug":"large-language-models-for-power-scheduling-a","title":"Large Language Models for Power Scheduling: A User-Centric Approach","date":"2024-06-29","arxiv_id":"2407.00476","repositories_listed":1,"syntology":null},{"url":"/paper/on-discrete-prompt-optimization-for-diffusion","slug":"on-discrete-prompt-optimization-for-diffusion","title":"On Discrete Prompt Optimization for Diffusion Models","date":"2024-06-27","arxiv_id":"2407.01606","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-discrete-prompt-optimization-for-diffusion#ran","syntology_url":"https://syntology.ai/paper/2407.01606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01606"}},"official":{"repos":["ruocwang/dpo-diffusion"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/assertionbench-a-benchmark-to-evaluate-large","slug":"assertionbench-a-benchmark-to-evaluate-large","title":"AssertionBench: A Benchmark to Evaluate Large-Language Models for Assertion Generation","date":"2024-06-26","arxiv_id":"2406.18627","repositories_listed":1,"syntology":null},{"url":"/paper/factfinders-at-checkthat-2024-refining-check","slug":"factfinders-at-checkthat-2024-refining-check","title":"FactFinders at CheckThat! 2024: Refining Check-worthy Statement Detection with LLMs through Data Pruning","date":"2024-06-26","arxiv_id":"2406.18297","repositories_listed":1,"syntology":null},{"url":"/paper/ctbench-a-comprehensive-benchmark-for","slug":"ctbench-a-comprehensive-benchmark-for","title":"CTBench: A Comprehensive Benchmark for Evaluating Language Model Capabilities in Clinical Trial Design","date":"2024-06-25","arxiv_id":"2406.17888","repositories_listed":1,"syntology":null},{"url":"/paper/feature-prompting-gbmseg-one-shot-reference","slug":"feature-prompting-gbmseg-one-shot-reference","title":"Feature-prompting GBMSeg: One-Shot Reference Guided Training-Free Prompt Engineering for Glomerular Basement Membrane Segmentation","date":"2024-06-24","arxiv_id":"2406.16271","repositories_listed":1,"syntology":null},{"url":"/paper/v-recs-a-low-cost-llm4vis-recommender-with","slug":"v-recs-a-low-cost-llm4vis-recommender-with","title":"V-RECS, a Low-Cost LLM4VIS Recommender with Explanations, Captioning and Suggestions","date":"2024-06-21","arxiv_id":"2406.15259","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-prompting-taxonomy-a-universal","slug":"hierarchical-prompting-taxonomy-a-universal","title":"Hierarchical Prompting Taxonomy: A Universal Evaluation Framework for Large Language Models Aligned with Human Cognitive Principles","date":"2024-06-18","arxiv_id":"2406.12644","repositories_listed":1,"syntology":null},{"url":"/paper/what-did-i-do-wrong-quantifying-llms","slug":"what-did-i-do-wrong-quantifying-llms","title":"What Did I Do Wrong? Quantifying LLMs' Sensitivity and Consistency to Prompt Engineering","date":"2024-06-18","arxiv_id":"2406.12334","repositories_listed":1,"syntology":null},{"url":"/paper/gaugllm-improving-graph-contrastive-learning","slug":"gaugllm-improving-graph-contrastive-learning","title":"GAugLLM: Improving Graph Contrastive Learning for Text-Attributed Graphs with Large Language Models","date":"2024-06-17","arxiv_id":"2406.11945","repositories_listed":1,"syntology":null},{"url":"/paper/grade-score-quantifying-llm-performance-in","slug":"grade-score-quantifying-llm-performance-in","title":"Grade Score: Quantifying LLM Performance in Option Selection","date":"2024-06-17","arxiv_id":"2406.12043","repositories_listed":1,"syntology":null},{"url":"/paper/self-reflection-outcome-is-sensitive-to","slug":"self-reflection-outcome-is-sensitive-to","title":"Self-Reflection Outcome is Sensitive to Prompt Construction","date":"2024-06-14","arxiv_id":"2406.10400","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-reflection-outcome-is-sensitive-to#ran","syntology_url":"https://syntology.ai/paper/2406.10400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10400"}},"official":{"repos":["michael98liu/mixture-of-prompts"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-scrutiny-detecting-backdoor-attacks","slug":"chain-of-scrutiny-detecting-backdoor-attacks","title":"Chain-of-Scrutiny: Detecting Backdoor Attacks for Large Language Models","date":"2024-06-10","arxiv_id":"2406.05948","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/chain-of-scrutiny-detecting-backdoor-attacks#ran","syntology_url":"https://syntology.ai/paper/2406.05948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05948"}},"official":{"repos":["lixi1994/CoS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/planning-like-human-a-dual-process-framework","slug":"planning-like-human-a-dual-process-framework","title":"Planning Like Human: A Dual-process Framework for Dialogue Planning","date":"2024-06-08","arxiv_id":"2406.05374","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/planning-like-human-a-dual-process-framework#ran","syntology_url":"https://syntology.ai/paper/2406.05374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05374"}},"official":{"repos":["cs-holder/DPDP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/online-joint-fine-tuning-of-multi-agent-flows","slug":"online-joint-fine-tuning-of-multi-agent-flows","title":"Online Joint Fine-tuning of Multi-Agent Flows","date":"2024-06-06","arxiv_id":"2406.04516","repositories_listed":1,"syntology":null},{"url":"/paper/pace-parsimonious-concept-engineering-for","slug":"pace-parsimonious-concept-engineering-for","title":"PaCE: Parsimonious Concept Engineering for Large Language Models","date":"2024-06-06","arxiv_id":"2406.04331","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pace-parsimonious-concept-engineering-for#ran","syntology_url":"https://syntology.ai/paper/2406.04331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04331"}},"official":{"repos":["peterljq/parsimonious-concept-engineering"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/verilogreader-llm-aided-hardware-test","slug":"verilogreader-llm-aided-hardware-test","title":"VerilogReader: LLM-Aided Hardware Test Generation","date":"2024-06-03","arxiv_id":"2406.04373","repositories_listed":1,"syntology":null},{"url":"/paper/interpretabnet-distilling-predictive-signals","slug":"interpretabnet-distilling-predictive-signals","title":"InterpreTabNet: Distilling Predictive Signals from Tabular Data by Salient Feature Interpretation","date":"2024-06-01","arxiv_id":"2406.00426","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":10,"n_ran_checked":10,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 10 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified; every one of the 10 samples that ran constructed an object rather than computing a result","sample_list":"/paper/interpretabnet-distilling-predictive-signals#ran","syntology_url":"https://syntology.ai/paper/2406.00426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00426"}},"official":{"repos":["jacobyhsi/InterpreTabNet"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":10,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/easy-problems-that-llms-get-wrong","slug":"easy-problems-that-llms-get-wrong","title":"Easy Problems That LLMs Get Wrong","date":"2024-05-30","arxiv_id":"2405.19616","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/easy-problems-that-llms-get-wrong#ran","syntology_url":"https://syntology.ai/paper/2405.19616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19616"}},"official":{"repos":["autogenai/easy-problems-that-llms-get-wrong"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-in-conversation-team-building-for","slug":"adaptive-in-conversation-team-building-for","title":"Adaptive In-conversation Team Building for Language Model Agents","date":"2024-05-29","arxiv_id":"2405.19425","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/adaptive-in-conversation-team-building-for#ran","syntology_url":"https://syntology.ai/paper/2405.19425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19425"}},"official":{"repos":["ag2ai/ag2"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/can-graph-learning-improve-task-planning","slug":"can-graph-learning-improve-task-planning","title":"Can Graph Learning Improve Planning in LLM-based Agents?","date":"2024-05-29","arxiv_id":"2405.19119","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-graph-learning-improve-task-planning#ran","syntology_url":"https://syntology.ai/paper/2405.19119","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19119"}},"official":{"repos":["wxxshirley/gnn4taskplan"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/orlm-training-large-language-models-for","slug":"orlm-training-large-language-models-for","title":"ORLM: A Customizable Framework in Training Large Models for Automated Optimization Modeling","date":"2024-05-28","arxiv_id":"2405.17743","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/orlm-training-large-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2405.17743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17743"}},"official":{"repos":["cardinal-operations/orlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/promptwizard-task-aware-agent-driven-prompt","slug":"promptwizard-task-aware-agent-driven-prompt","title":"PromptWizard: Task-Aware Prompt Optimization Framework","date":"2024-05-28","arxiv_id":"2405.18369","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/promptwizard-task-aware-agent-driven-prompt#ran","syntology_url":"https://syntology.ai/paper/2405.18369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18369"}},"official":null}},{"url":"/paper/adapting-promptore-for-modern-history","slug":"adapting-promptore-for-modern-history","title":"Adapting PromptORE for Modern History: Information Extraction from Hispanic Monarchy Documents of the XVIth Century","date":"2024-05-24","arxiv_id":"2406.00027","repositories_listed":1,"syntology":null},{"url":"/paper/what-do-you-see-enhancing-zero-shot-image","slug":"what-do-you-see-enhancing-zero-shot-image","title":"What Do You See? Enhancing Zero-Shot Image Classification with Multimodal Large Language Models","date":"2024-05-24","arxiv_id":"2405.15668","repositories_listed":1,"syntology":null},{"url":"/paper/a-lost-opportunity-for-vision-language-models","slug":"a-lost-opportunity-for-vision-language-models","title":"A Lost Opportunity for Vision-Language Models: A Comparative Study of Online Test-Time Adaptation for Vision-Language Models","date":"2024-05-23","arxiv_id":"2405.14977","repositories_listed":1,"syntology":null},{"url":"/paper/e2tp-element-to-tuple-prompting-improves","slug":"e2tp-element-to-tuple-prompting-improves","title":"E2TP: Element to Tuple Prompting Improves Aspect Sentiment Tuple Prediction","date":"2024-05-10","arxiv_id":"2405.06454","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-capabilities-of-large","slug":"exploring-the-capabilities-of-large","title":"Exploring the Capabilities of Large Multimodal Models on Dense Text","date":"2024-05-09","arxiv_id":"2405.06706","repositories_listed":1,"syntology":null},{"url":"/paper/cityllava-efficient-fine-tuning-for-vlms-in","slug":"cityllava-efficient-fine-tuning-for-vlms-in","title":"CityLLaVA: Efficient Fine-Tuning for VLMs in City Scenario","date":"2024-05-06","arxiv_id":"2405.03194","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cityllava-efficient-fine-tuning-for-vlms-in#ran","syntology_url":"https://syntology.ai/paper/2405.03194","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.03194"}},"official":{"repos":["alibaba/aicity2024_track2_aliopentrek_cityllava"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-a-human-in-the-loop-llm-approach-to","slug":"towards-a-human-in-the-loop-llm-approach-to","title":"Towards A Human-in-the-Loop LLM Approach to Collaborative Discourse Analysis","date":"2024-05-06","arxiv_id":"2405.03677","repositories_listed":1,"syntology":null},{"url":"/paper/aloe-a-family-of-fine-tuned-open-healthcare","slug":"aloe-a-family-of-fine-tuned-open-healthcare","title":"Aloe: A Family of Fine-tuned Open Healthcare LLMs","date":"2024-05-03","arxiv_id":"2405.01886","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/aloe-a-family-of-fine-tuned-open-healthcare#ran","syntology_url":"https://syntology.ai/paper/2405.01886","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.01886"}},"official":null}},{"url":"/paper/cactus-chemistry-agent-connecting-tool-usage","slug":"cactus-chemistry-agent-connecting-tool-usage","title":"CACTUS: Chemistry Agent Connecting Tool-Usage to Science","date":"2024-05-02","arxiv_id":"2405.00972","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-capabilities-of-large-language","slug":"exploring-the-capabilities-of-large-language","title":"Exploring the Capabilities of Large Language Models for Generating Diverse Design Solutions","date":"2024-05-02","arxiv_id":"2405.02345","repositories_listed":1,"syntology":null},{"url":"/paper/wiba-what-is-being-argued-a-comprehensive","slug":"wiba-what-is-being-argued-a-comprehensive","title":"WIBA: What Is Being Argued? A Comprehensive Approach to Argument Mining","date":"2024-05-01","arxiv_id":"2405.00828","repositories_listed":1,"syntology":null},{"url":"/paper/umass-bionlp-at-mediqa-m3g-2024-dermprompt-a","slug":"umass-bionlp-at-mediqa-m3g-2024-dermprompt-a","title":"UMass-BioNLP at MEDIQA-M3G 2024: DermPrompt -- A Systematic Exploration of Prompt Engineering with GPT-4V for Dermatological Diagnosis","date":"2024-04-27","arxiv_id":"2404.17749","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistic-inference-in-language-models","slug":"probabilistic-inference-in-language-models","title":"Probabilistic Inference in Language Models via Twisted Sequential Monte Carlo","date":"2024-04-26","arxiv_id":"2404.17546","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":17,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/probabilistic-inference-in-language-models#ran","syntology_url":"https://syntology.ai/paper/2404.17546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.17546"}},"official":{"repos":["silent-zebra/twisted-smc-lm"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/integrating-chemistry-knowledge-in-large","slug":"integrating-chemistry-knowledge-in-large","title":"Integrating Chemistry Knowledge in Large Language Models via Prompt Engineering","date":"2024-04-22","arxiv_id":"2404.14467","repositories_listed":1,"syntology":null},{"url":"/paper/sample-design-engineering-an-empirical-study","slug":"sample-design-engineering-an-empirical-study","title":"Sample Design Engineering: An Empirical Study of What Makes Good Downstream Fine-Tuning Samples for LLMs","date":"2024-04-19","arxiv_id":"2404.13033","repositories_listed":1,"syntology":null},{"url":"/paper/towards-reliable-latent-knowledge-estimation","slug":"towards-reliable-latent-knowledge-estimation","title":"Towards Reliable Latent Knowledge Estimation in LLMs: Zero-Prompt Many-Shot Based Factual Knowledge Extraction","date":"2024-04-19","arxiv_id":"2404.12957","repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-and-llm-based-methods-applied","slug":"deep-learning-and-llm-based-methods-applied","title":"Deep Learning and LLM-based Methods Applied to Stellar Lightcurve Classification","date":"2024-04-16","arxiv_id":"2404.10757","repositories_listed":1,"syntology":null},{"url":"/paper/incubating-text-classifiers-following-user","slug":"incubating-text-classifiers-following-user","title":"Incubating Text Classifiers Following User Instruction with Nothing but LLM","date":"2024-04-16","arxiv_id":"2404.10877","repositories_listed":1,"syntology":null},{"url":"/paper/rumour-evaluation-with-very-large-language","slug":"rumour-evaluation-with-very-large-language","title":"Rumour Evaluation with Very Large Language Models","date":"2024-04-11","arxiv_id":"2404.16859","repositories_listed":1,"syntology":null},{"url":"/paper/test-time-adaptation-with-salip-a-cascade-of","slug":"test-time-adaptation-with-salip-a-cascade-of","title":"Test-Time Adaptation with SaLIP: A Cascade of SAM and CLIP for Zero shot Medical Image Segmentation","date":"2024-04-09","arxiv_id":"2404.06362","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/test-time-adaptation-with-salip-a-cascade-of#ran","syntology_url":"https://syntology.ai/paper/2404.06362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.06362"}},"official":{"repos":["aleemsidra/SaLIP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/promptad-learning-prompts-with-only-normal","slug":"promptad-learning-prompts-with-only-normal","title":"PromptAD: Learning Prompts with only Normal Samples for Few-Shot Anomaly Detection","date":"2024-04-08","arxiv_id":"2404.05231","repositories_listed":1,"syntology":{"n":16,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":10,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/promptad-learning-prompts-with-only-normal#ran","syntology_url":"https://syntology.ai/paper/2404.05231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05231"}},"official":{"repos":["funz-0/promptad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/utebc-nlp-at-semeval-2024-task-9-can-llms-be","slug":"utebc-nlp-at-semeval-2024-task-9-can-llms-be","title":"uTeBC-NLP at SemEval-2024 Task 9: Can LLMs be Lateral Thinkers?","date":"2024-04-03","arxiv_id":"2404.02474","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/utebc-nlp-at-semeval-2024-task-9-can-llms-be#ran","syntology_url":"https://syntology.ai/paper/2404.02474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02474"}},"official":{"repos":["ipouyall/can-llms-be-lateral-thinkers"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompt-learning-via-meta-regularization","slug":"prompt-learning-via-meta-regularization","title":"Prompt Learning via Meta-Regularization","date":"2024-04-01","arxiv_id":"2404.00851","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prompt-learning-via-meta-regularization#ran","syntology_url":"https://syntology.ai/paper/2404.00851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00851"}},"official":{"repos":["mlvlab/prometar"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-are-spacecraft-operators","slug":"language-models-are-spacecraft-operators","title":"Language Models are Spacecraft Operators","date":"2024-03-30","arxiv_id":"2404.00413","repositories_listed":1,"syntology":null},{"url":"/paper/enhanced-short-text-modeling-leveraging-large","slug":"enhanced-short-text-modeling-leveraging-large","title":"Enhanced Short Text Modeling: Leveraging Large Language Models for Topic Refinement","date":"2024-03-26","arxiv_id":"2403.17706","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparison-of-human-gpt-3-5-and-gpt-4","slug":"a-comparison-of-human-gpt-3-5-and-gpt-4","title":"A comparison of Human, GPT-3.5, and GPT-4 Performance in a University-Level Coding Course","date":"2024-03-25","arxiv_id":"2403.16977","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-impact-of-the-output-format-on","slug":"exploring-the-impact-of-the-output-format-on","title":"Exploring the Impact of the Output Format on the Evaluation of Large Language Models for Code Translation","date":"2024-03-25","arxiv_id":"2403.17214","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-large-language-model-to-generate-a","slug":"leveraging-large-language-model-to-generate-a","title":"Leveraging Large Language Model to Generate a Novel Metaheuristic Algorithm with CRISPE Framework","date":"2024-03-25","arxiv_id":"2403.16417","repositories_listed":1,"syntology":null},{"url":"/paper/linear-cross-document-event-coreference","slug":"linear-cross-document-event-coreference","title":"Linear Cross-document Event Coreference Resolution with X-AMR","date":"2024-03-25","arxiv_id":"2404.08656","repositories_listed":1,"syntology":null},{"url":"/paper/lamper-language-model-and-prompt-engineering","slug":"lamper-language-model-and-prompt-engineering","title":"LAMPER: LanguAge Model and Prompt EngineeRing for zero-shot time series classification","date":"2024-03-23","arxiv_id":"2403.15875","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/lamper-language-model-and-prompt-engineering#ran","syntology_url":"https://syntology.ai/paper/2403.15875","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15875"}},"official":{"repos":["dodoxxb/lamper"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/comprehensive-evaluation-and-insights-into-1","slug":"comprehensive-evaluation-and-insights-into-1","title":"Comprehensive Evaluation and Insights into the Use of Large Language Models in the Automation of Behavior-Driven Development Acceptance Test Formulation","date":"2024-03-22","arxiv_id":"2403.14965","repositories_listed":1,"syntology":null},{"url":"/paper/instasynth-opportunities-and-challenges-in","slug":"instasynth-opportunities-and-challenges-in","title":"InstaSynth: Opportunities and Challenges in Generating Synthetic Instagram Data with ChatGPT for Sponsored Content Detection","date":"2024-03-22","arxiv_id":"2403.15214","repositories_listed":1,"syntology":null},{"url":"/paper/can-chatgpt-detect-deepfakes-a-study-of-using","slug":"can-chatgpt-detect-deepfakes-a-study-of-using","title":"Can ChatGPT Detect DeepFakes? A Study of Using Multimodal Large Language Models for Media Forensics","date":"2024-03-21","arxiv_id":"2403.14077","repositories_listed":1,"syntology":null},{"url":"/paper/defending-against-indirect-prompt-injection","slug":"defending-against-indirect-prompt-injection","title":"Defending Against Indirect Prompt Injection Attacks With Spotlighting","date":"2024-03-20","arxiv_id":"2403.14720","repositories_listed":1,"syntology":null},{"url":"/paper/just-shift-it-test-time-prototype-shifting","slug":"just-shift-it-test-time-prototype-shifting","title":"Just Shift It: Test-Time Prototype Shifting for Zero-Shot Generalization with Vision-Language Models","date":"2024-03-19","arxiv_id":"2403.12952","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/just-shift-it-test-time-prototype-shifting#ran","syntology_url":"https://syntology.ai/paper/2403.12952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12952"}},"official":{"repos":["elaine-sui/tps"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dialoggen-multi-modal-interactive-dialogue","slug":"dialoggen-multi-modal-interactive-dialogue","title":"DialogGen: Multi-modal Interactive Dialogue System for Multi-turn Text-to-Image Generation","date":"2024-03-13","arxiv_id":"2403.08857","repositories_listed":1,"syntology":null},{"url":"/paper/can-large-language-models-automatically-score","slug":"can-large-language-models-automatically-score","title":"Can Large Language Models Automatically Score Proficiency of Written Essays?","date":"2024-03-10","arxiv_id":"2403.06149","repositories_listed":1,"syntology":null},{"url":"/paper/vidprom-a-million-scale-real-prompt-gallery","slug":"vidprom-a-million-scale-real-prompt-gallery","title":"VidProM: A Million-scale Real Prompt-Gallery Dataset for Text-to-Video Diffusion Models","date":"2024-03-10","arxiv_id":"2403.06098","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vidprom-a-million-scale-real-prompt-gallery#ran","syntology_url":"https://syntology.ai/paper/2403.06098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.06098"}},"official":{"repos":["wangwenhao0716/vidprom"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/erbench-an-entity-relationship-based","slug":"erbench-an-entity-relationship-based","title":"ERBench: An Entity-Relationship based Automatically Verifiable Hallucination Benchmark for Large Language Models","date":"2024-03-08","arxiv_id":"2403.05266","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/erbench-an-entity-relationship-based#ran","syntology_url":"https://syntology.ai/paper/2403.05266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05266"}},"official":{"repos":["dilab-kaist/erbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/promptcharm-text-to-image-generation-through","slug":"promptcharm-text-to-image-generation-through","title":"PromptCharm: Text-to-Image Generation through Multi-modal Prompting and Refinement","date":"2024-03-06","arxiv_id":"2403.04014","repositories_listed":1,"syntology":null},{"url":"/paper/promptkd-unsupervised-prompt-distillation-for","slug":"promptkd-unsupervised-prompt-distillation-for","title":"PromptKD: Unsupervised Prompt Distillation for Vision-Language Models","date":"2024-03-05","arxiv_id":"2403.02781","repositories_listed":1,"syntology":null},{"url":"/paper/offlandat-a-community-based-implicit","slug":"offlandat-a-community-based-implicit","title":"OffensiveLang: A Community Based Implicit Offensive Language Dataset","date":"2024-03-04","arxiv_id":"2403.02472","repositories_listed":1,"syntology":null},{"url":"/paper/cogbench-a-large-language-model-walks-into-a","slug":"cogbench-a-large-language-model-walks-into-a","title":"CogBench: a large language model walks into a psychology lab","date":"2024-02-28","arxiv_id":"2402.18225","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cogbench-a-large-language-model-walks-into-a#ran","syntology_url":"https://syntology.ai/paper/2402.18225","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18225"}},"official":{"repos":["juliancodaforno/cogbench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-pro-learning-to-evolve-via-policy-level","slug":"agent-pro-learning-to-evolve-via-policy-level","title":"Agent-Pro: Learning to Evolve via Policy-Level Reflection and Optimization","date":"2024-02-27","arxiv_id":"2402.17574","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agent-pro-learning-to-evolve-via-policy-level#ran","syntology_url":"https://syntology.ai/paper/2402.17574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17574"}},"official":{"repos":["zwq2018/agent-pro"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/parameter-efficient-prompt-learning-for-3d","slug":"parameter-efficient-prompt-learning-for-3d","title":"Parameter-efficient Prompt Learning for 3D Point Cloud Understanding","date":"2024-02-24","arxiv_id":"2402.15823","repositories_listed":1,"syntology":null},{"url":"/paper/llm-based-multi-agent-generation-of-semi","slug":"llm-based-multi-agent-generation-of-semi","title":"LLM Based Multi-Agent Generation of Semi-structured Documents from Semantic Templates in the Public Administration Domain","date":"2024-02-21","arxiv_id":"2402.14871","repositories_listed":1,"syntology":null},{"url":"/paper/a-user-friendly-framework-for-generating","slug":"a-user-friendly-framework-for-generating","title":"A User-Friendly Framework for Generating Model-Preferred Prompts in Text-to-Image Synthesis","date":"2024-02-20","arxiv_id":"2402.12760","repositories_listed":1,"syntology":null},{"url":"/paper/how-interpretable-are-reasoning-explanations","slug":"how-interpretable-are-reasoning-explanations","title":"How Interpretable are Reasoning Explanations from Prompting Large Language Models?","date":"2024-02-19","arxiv_id":"2402.11863","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":16,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/how-interpretable-are-reasoning-explanations#ran","syntology_url":"https://syntology.ai/paper/2402.11863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11863"}},"official":{"repos":["wj210/cot_interpretability"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}}],"record_sha256":"72ec8d11b1c933469ba58785e18220ae0a7e60c17464da9b377820d6da8cfc0e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}