{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/14","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":14,"pages_in_order":29,"rows_per_page":100,"rows":[1301,1400],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/13","next":"/method/gpt-4/papers/15","papers":[{"paper":"/paper/humor-mechanics-advancing-humor-generation","slug":"humor-mechanics-advancing-humor-generation","title":"Humor Mechanics: Advancing Humor Generation with Multistep Reasoning","date":"2024-05-12","arxiv_id":"2405.07280","n_code_links":1,"syntology":null},{"paper":null,"slug":"l-u-pin-llm-based-political-ideology","title":"L(u)PIN: LLM-based Political Ideology Nowcasting","date":"2024-05-12","arxiv_id":"2405.07320","n_code_links":0,"syntology":null},{"paper":"/paper/limited-ability-of-llms-to-simulate-human","slug":"limited-ability-of-llms-to-simulate-human","title":"Limited Ability of LLMs to Simulate Human Psychological Behaviours: a Psychometric Analysis","date":"2024-05-12","arxiv_id":"2405.07248","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nikbpetrov/llms-simulate-humans"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/medconceptsqa-open-source-medical-concepts-qa","slug":"medconceptsqa-open-source-medical-concepts-qa","title":"MedConceptsQA: Open Source Medical Concepts QA Benchmark","date":"2024-05-12","arxiv_id":"2405.07348","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nadavlab/MedConceptsQA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"automating-thematic-analysis-how-llms-analyse","title":"Automating Thematic Analysis: How LLMs Analyse Controversial Topics","date":"2024-05-11","arxiv_id":"2405.06919","n_code_links":0,"syntology":null},{"paper":null,"slug":"identifying-key-terms-in-prompts-for","title":"Identifying Key Terms in Prompts for Relevance Evaluation with GPT Models","date":"2024-05-11","arxiv_id":"2405.06931","n_code_links":0,"syntology":null},{"paper":null,"slug":"tacoere-cluster-aware-compression-for-event","title":"TacoERE: Cluster-aware Compression for Event Relation Extraction","date":"2024-05-11","arxiv_id":"2405.06890","n_code_links":0,"syntology":null},{"paper":null,"slug":"canal-cyber-activity-news-alerting-language","title":"CANAL -- Cyber Activity News Alerting Language Model: Empirical Approach vs. Expensive LLM","date":"2024-05-10","arxiv_id":"2405.06772","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-model-in-financial-regulatory","title":"Large Language Model in Financial Regulatory Interpretation","date":"2024-05-10","arxiv_id":"2405.06808","n_code_links":0,"syntology":null},{"paper":"/paper/multimodal-llms-struggle-with-basic-visual","slug":"multimodal-llms-struggle-with-basic-visual","title":"Multimodal LLMs Struggle with Basic Visual Network Analysis: a VNA Benchmark","date":"2024-05-10","arxiv_id":"2405.06634","n_code_links":1,"syntology":null},{"paper":null,"slug":"artificial-intelligence-as-the-new-hacker","title":"Artificial Intelligence as the New Hacker: Developing Agents for Offensive Security","date":"2024-05-09","arxiv_id":"2406.07561","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-large-language-models-understand-uncommon","title":"Can large language models understand uncommon meanings of common words?","date":"2024-05-09","arxiv_id":"2405.05741","n_code_links":0,"syntology":null},{"paper":null,"slug":"digital-diagnostics-the-potential-of-large","title":"Digital Diagnostics: The Potential Of Large Language Models In Recognizing Symptoms Of Common Illnesses","date":"2024-05-09","arxiv_id":"2405.06712","n_code_links":0,"syntology":null},{"paper":null,"slug":"letter-to-the-editor-what-are-the-legal-and","title":"Letter to the Editor: What are the legal and ethical considerations of submitting radiology reports to ChatGPT?","date":"2024-05-09","arxiv_id":"2405.05647","n_code_links":0,"syntology":null},{"paper":null,"slug":"people-cannot-distinguish-gpt-4-from-a-human","title":"People cannot distinguish GPT-4 from a human in a Turing test","date":"2024-05-09","arxiv_id":"2405.08007","n_code_links":0,"syntology":null},{"paper":"/paper/smurfs-leveraging-multiple-proficiency-agents","slug":"smurfs-leveraging-multiple-proficiency-agents","title":"Smurfs: Leveraging Multiple Proficiency Agents with Context-Efficiency for Tool Planning","date":"2024-05-09","arxiv_id":"2405.05955","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["freedomintelligence/smurfs"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/acorn-aspect-wise-commonsense-reasoning","slug":"acorn-aspect-wise-commonsense-reasoning","title":"ACORN: Aspect-wise Commonsense Reasoning Explanation Evaluation","date":"2024-05-08","arxiv_id":"2405.04818","n_code_links":1,"syntology":null},{"paper":"/paper/open-source-language-models-can-provide","slug":"open-source-language-models-can-provide","title":"Open Source Language Models Can Provide Feedback: Evaluating LLMs' Ability to Help Students Using GPT-4-As-A-Judge","date":"2024-05-08","arxiv_id":"2405.05253","n_code_links":1,"syntology":null},{"paper":null,"slug":"seeds-of-stereotypes-a-large-scale-textual","title":"Seeds of Stereotypes: A Large-Scale Textual Analysis of Race and Gender Associations with Diseases in Online Sources","date":"2024-05-08","arxiv_id":"2405.05049","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-the-efficiency-and-accuracy-of","slug":"enhancing-the-efficiency-and-accuracy-of","title":"Enhancing the Efficiency and Accuracy of Underlying Asset Reviews in Structured Finance: The Application of Multi-agent Framework","date":"2024-05-07","arxiv_id":"2405.04294","n_code_links":1,"syntology":null},{"paper":"/paper/naturalcodebench-examining-coding-performance","slug":"naturalcodebench-examining-coding-performance","title":"NaturalCodeBench: Examining Coding Performance Mismatch on HumanEval and Natural User Prompts","date":"2024-05-07","arxiv_id":"2405.04520","n_code_links":1,"syntology":null},{"paper":"/paper/alphamath-almost-zero-process-supervision","slug":"alphamath-almost-zero-process-supervision","title":"AlphaMath Almost Zero: Process Supervision without Process","date":"2024-05-06","arxiv_id":"2405.03553","n_code_links":1,"syntology":null},{"paper":"/paper/anchored-answers-unravelling-positional-bias","slug":"anchored-answers-unravelling-positional-bias","title":"Anchored Answers: Unravelling Positional Bias in GPT-2's Multiple-Choice Questions","date":"2024-05-06","arxiv_id":"2405.03205","n_code_links":1,"syntology":null},{"paper":null,"slug":"green-generative-radiology-report-evaluation","title":"GREEN: Generative Radiology Report Evaluation and Error Notation","date":"2024-05-06","arxiv_id":"2405.03595","n_code_links":0,"syntology":null},{"paper":null,"slug":"mammoth2-scaling-instructions-from-the-web","title":"MAmmoTH2: Scaling Instructions from the Web","date":"2024-05-06","arxiv_id":"2405.03548","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-large-language-models-make-the-grade-an","title":"Can Large Language Models Make the Grade? An Empirical Study Evaluating LLMs Ability to Mark Short Answer Questions in K-12 Education","date":"2024-05-05","arxiv_id":"2405.02985","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-lecture-content-for-improved","title":"Leveraging Lecture Content for Improved Feedback: Explorations with GPT-4 and Retrieval Augmented Generation","date":"2024-05-05","arxiv_id":"2405.06681","n_code_links":0,"syntology":null},{"paper":"/paper/negativeprompt-leveraging-psychology-for","slug":"negativeprompt-leveraging-psychology-for","title":"NegativePrompt: Leveraging Psychology for Large Language Models Enhancement via Negative Emotional Stimuli","date":"2024-05-05","arxiv_id":"2405.02814","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wangxu0820/negativeprompt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"open-sql-framework-enhancing-text-to-sql-on","title":"Open-SQL Framework: Enhancing Text-to-SQL on Open-source Large Language Models","date":"2024-05-04","arxiv_id":"2405.06674","n_code_links":0,"syntology":null},{"paper":"/paper/propertygpt-llm-driven-formal-verification-of","slug":"propertygpt-llm-driven-formal-verification-of","title":"PropertyGPT: LLM-driven Formal Verification of Smart Contracts through Retrieval-Augmented Property Generation","date":"2024-05-04","arxiv_id":"2405.02580","n_code_links":1,"syntology":null},{"paper":"/paper/automating-the-enterprise-with-foundation","slug":"automating-the-enterprise-with-foundation","title":"Automating the Enterprise with Foundation Models","date":"2024-05-03","arxiv_id":"2405.03710","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hazyresearch/eclair-agents"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"comparative-analysis-of-retrieval-systems-in","title":"Comparative Analysis of Retrieval Systems in the Real World","date":"2024-05-03","arxiv_id":"2405.02048","n_code_links":0,"syntology":null},{"paper":null,"slug":"reasons-a-benchmark-for-retrieval-and","title":"Attribution in Scientific Literature: New Benchmark and Methods","date":"2024-05-03","arxiv_id":"2405.02228","n_code_links":0,"syntology":null},{"paper":"/paper/single-and-multi-hop-question-answering","slug":"single-and-multi-hop-question-answering","title":"Single and Multi-Hop Question-Answering Datasets for Reticular Chemistry with GPT-4-Turbo","date":"2024-05-03","arxiv_id":"2405.02128","n_code_links":1,"syntology":null},{"paper":"/paper/a-survey-on-large-language-models-for-3","slug":"a-survey-on-large-language-models-for-3","title":"A Survey on Large Language Models for Critical Societal Domains: Finance, Healthcare, and Law","date":"2024-05-02","arxiv_id":"2405.01769","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-can-i-get-it-right-using-gpt-to-rephrase","title":"How Can I Get It Right? Using GPT to Rephrase Incorrect Trainee Responses","date":"2024-05-02","arxiv_id":"2405.00970","n_code_links":0,"syntology":null},{"paper":"/paper/minigpt-3d-efficiently-aligning-3d-point","slug":"minigpt-3d-efficiently-aligning-3d-point","title":"MiniGPT-3D: Efficiently Aligning 3D Point Clouds with Large Language Models using 2D Priors","date":"2024-05-02","arxiv_id":"2405.01413","n_code_links":1,"syntology":null},{"paper":"/paper/prometheus-2-an-open-source-language-model","slug":"prometheus-2-an-open-source-language-model","title":"Prometheus 2: An Open Source Language Model Specialized in Evaluating Other Language Models","date":"2024-05-02","arxiv_id":"2405.01535","n_code_links":1,"syntology":null},{"paper":null,"slug":"wildchat-1m-chatgpt-interaction-logs-in-the","title":"WildChat: 1M ChatGPT Interaction Logs in the Wild","date":"2024-05-02","arxiv_id":"2405.01470","n_code_links":0,"syntology":null},{"paper":null,"slug":"courseassist-pedagogically-appropriate","title":"CourseAssist: Pedagogically Appropriate AI Tutor for Computer Science Education","date":"2024-05-01","arxiv_id":"2407.10246","n_code_links":0,"syntology":null},{"paper":"/paper/self-play-preference-optimization-for","slug":"self-play-preference-optimization-for","title":"Self-Play Preference Optimization for Language Model Alignment","date":"2024-05-01","arxiv_id":"2405.00675","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["uclaml/sppo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-framework-for-leveraging-human-computation","title":"A Framework for Leveraging Human Computation Gaming to Enhance Knowledge Graphs for Accuracy Critical Generative AI Applications","date":"2024-04-30","arxiv_id":"2404.19729","n_code_links":0,"syntology":null},{"paper":"/paper/constrained-decoding-for-secure-code","slug":"constrained-decoding-for-secure-code","title":"Constrained Decoding for Secure Code Generation","date":"2024-04-30","arxiv_id":"2405.00218","n_code_links":2,"syntology":{"ran":9,"of":13,"n_ran_checked":9,"n_instrument":0,"unverified":4,"pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["dynamite321/codeguardplus"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/do-large-language-models-understand","slug":"do-large-language-models-understand","title":"Do Large Language Models Understand Conversational Implicature -- A case study with a chinese sitcom","date":"2024-04-30","arxiv_id":"2404.19509","n_code_links":1,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["sjtu-compling/llm-pragmatics"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/extending-llama-3-s-context-ten-fold","slug":"extending-llama-3-s-context-ten-fold","title":"Extending Llama-3's Context Ten-Fold Overnight","date":"2024-04-30","arxiv_id":"2404.19553","n_code_links":1,"syntology":null},{"paper":null,"slug":"harmonic-llms-are-trustworthy","title":"Harmonic LLMs are Trustworthy","date":"2024-04-30","arxiv_id":"2404.19708","n_code_links":0,"syntology":null},{"paper":null,"slug":"octopus-v4-graph-of-language-models","title":"Octopus v4: Graph of language models","date":"2024-04-30","arxiv_id":"2404.19296","n_code_links":0,"syntology":null},{"paper":"/paper/repeval-effective-text-evaluation-with-llm","slug":"repeval-effective-text-evaluation-with-llm","title":"RepEval: Effective Text Evaluation with LLM Representation","date":"2024-04-30","arxiv_id":"2404.19563","n_code_links":1,"syntology":{"ran":18,"of":27,"n_ran_checked":15,"n_instrument":3,"unverified":9,"pointer_only":6,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 2 honoured, 0 violated, 13 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","official":{"repos":["susisheng/repeval"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":null,"slug":"automated-construction-of-theme-specific","title":"Automated Construction of Theme-specific Knowledge Graphs","date":"2024-04-29","arxiv_id":"2404.19146","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-gpt-4-do-l2-analytic-assessment","title":"Can GPT-4 do L2 analytic assessment?","date":"2024-04-29","arxiv_id":"2404.18557","n_code_links":0,"syntology":null},{"paper":"/paper/capabilities-of-gemini-models-in-medicine","slug":"capabilities-of-gemini-models-in-medicine","title":"Capabilities of Gemini Models in Medicine","date":"2024-04-29","arxiv_id":"2404.18416","n_code_links":0,"syntology":null},{"paper":"/paper/do-neutral-prompts-produce-insecure-code","slug":"do-neutral-prompts-produce-insecure-code","title":"How secure is AI-generated Code: A Large-Scale Comparison of Large Language Models","date":"2024-04-29","arxiv_id":"2404.18353","n_code_links":1,"syntology":null},{"paper":null,"slug":"ethical-reasoning-and-moral-value-alignment","title":"Ethical Reasoning and Moral Value Alignment of LLMs Depend on the Language we Prompt them in","date":"2024-04-29","arxiv_id":"2404.18460","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-passes-most-of-the-297-written-polish","title":"GPT-4 passes most of the 297 written Polish Board Certification Examinations","date":"2024-04-29","arxiv_id":"2405.01589","n_code_links":0,"syntology":null},{"paper":"/paper/lora-land-310-fine-tuned-llms-that-rival-gpt","slug":"lora-land-310-fine-tuned-llms-that-rival-gpt","title":"LoRA Land: 310 Fine-tuned LLMs that Rival GPT-4, A Technical Report","date":"2024-04-29","arxiv_id":"2405.00732","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":11,"n_instrument":0,"unverified":1,"pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["predibase/lora_bakeoff"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/composerx-multi-agent-symbolic-music","slug":"composerx-multi-agent-symbolic-music","title":"ComposerX: Multi-Agent Symbolic Music Composition with LLMs","date":"2024-04-28","arxiv_id":"2404.18081","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["lllindsey0615/composerx"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"patentgpt-a-large-language-model-for","title":"PatentGPT: A Large Language Model for Intellectual Property","date":"2024-04-28","arxiv_id":"2404.18255","n_code_links":0,"syntology":null},{"paper":null,"slug":"advancing-healthcare-automation-multi-agent","title":"Advancing Healthcare Automation: Multi-Agent System for Medical Necessity Justification","date":"2024-04-27","arxiv_id":"2404.17977","n_code_links":0,"syntology":null},{"paper":null,"slug":"automating-customer-needs-analysis-a","title":"Automating Customer Needs Analysis: A Comparative Study of Large Language Models in the Travel Industry","date":"2024-04-27","arxiv_id":"2404.17975","n_code_links":0,"syntology":null},{"paper":null,"slug":"detection-of-conspiracy-theories-beyond","title":"Detection of Conspiracy Theories Beyond Keyword Bias in German-Language Telegram Using Large Language Models","date":"2024-04-27","arxiv_id":"2404.17985","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-pre-trained-generative-language","title":"Enhancing Pre-Trained Generative Language Models with Question Attended Span Extraction on Machine Reading Comprehension","date":"2024-04-27","arxiv_id":"2404.17991","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-few-shot-learning-for","title":"Evaluation of Few-Shot Learning for Classification Tasks in the Polish Language","date":"2024-04-27","arxiv_id":"2404.17832","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-llms-in-software-requirements","title":"Using LLMs in Software Requirements Specifications: An Empirical Evaluation","date":"2024-04-27","arxiv_id":"2404.17842","n_code_links":0,"syntology":null},{"paper":"/paper/adapting-open-source-large-language-models","slug":"adapting-open-source-large-language-models","title":"Towards Adapting Open-Source Large Language Models for Expert-Level Clinical Note Generation","date":"2024-04-25","arxiv_id":"2405.00715","n_code_links":1,"syntology":null},{"paper":"/paper/ai-coders-are-among-us-rethinking-programming","slug":"ai-coders-are-among-us-rethinking-programming","title":"AI Coders Are Among Us: Rethinking Programming Language Grammar Towards Efficient Code Generation","date":"2024-04-25","arxiv_id":"2404.16333","n_code_links":1,"syntology":null},{"paper":"/paper/global-concept-explanations-for-graphs-by","slug":"global-concept-explanations-for-graphs-by","title":"Global Concept Explanations for Graphs by Contrastive Learning","date":"2024-04-25","arxiv_id":"2404.16532","n_code_links":2,"syntology":null},{"paper":"/paper/indicgenbench-a-multilingual-benchmark-to","slug":"indicgenbench-a-multilingual-benchmark-to","title":"IndicGenBench: A Multilingual Benchmark to Evaluate Generation Capabilities of LLMs on Indic Languages","date":"2024-04-25","arxiv_id":"2404.16816","n_code_links":1,"syntology":null},{"paper":null,"slug":"influence-of-solution-efficiency-and-valence","title":"Influence of Solution Efficiency and Valence of Instruction on Additive and Subtractive Solution Strategies in Humans and GPT-4","date":"2024-04-25","arxiv_id":"2404.16692","n_code_links":0,"syntology":null},{"paper":"/paper/llm-based-section-identifiers-excel-on-open","slug":"llm-based-section-identifiers-excel-on-open","title":"LLM-Based Section Identifiers Excel on Open Source but Stumble in Real World Applications","date":"2024-04-25","arxiv_id":"2404.16294","n_code_links":1,"syntology":null},{"paper":null,"slug":"player-driven-emergence-in-llm-driven-game","title":"Player-Driven Emergence in LLM-Driven Game Narrative","date":"2024-04-25","arxiv_id":"2404.17027","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-gpt-surprise-offering-large-language","title":"The GPT Surprise: Offering Large Language Model Chat in a Massive Coding Class Reduced Engagement but Increased Adopters Exam Performances","date":"2024-04-25","arxiv_id":"2407.09975","n_code_links":0,"syntology":null},{"paper":null,"slug":"utilizing-large-language-models-to-identify","title":"Utilizing Large Language Models to Identify Reddit Users Considering Vaping Cessation for Digital Interventions","date":"2024-04-25","arxiv_id":"2404.17607","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-the-potential-of-mid-sized-language","title":"Assessing The Potential Of Mid-Sized Language Models For Clinical QA","date":"2024-04-24","arxiv_id":"2404.15894","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-foundational-large-language-models-assist","title":"Can Foundational Large Language Models Assist with Conducting Pharmaceuticals Manufacturing Investigations?","date":"2024-04-24","arxiv_id":"2404.15578","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-the-prompt-leakage-effect-and","title":"Prompt Leakage effect and defense strategies for multi-turn LLM interactions","date":"2024-04-24","arxiv_id":"2404.16251","n_code_links":0,"syntology":null},{"paper":"/paper/multi-modal-proxy-learning-towards","slug":"multi-modal-proxy-learning-towards","title":"Multi-Modal Proxy Learning Towards Personalized Visual Multiple Clustering","date":"2024-04-24","arxiv_id":"2404.15655","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":1,"n_instrument":3,"unverified":3,"pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["alexander-yao/multi-map"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-promise-and-challenges-of-using-llms-to","title":"The Promise and Challenges of Using LLMs to Accelerate the Screening Process of Systematic Reviews","date":"2024-04-24","arxiv_id":"2404.15667","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-a-holistic-evaluation-of-llms-on","title":"Towards a Holistic Evaluation of LLMs on Factual Knowledge Recall","date":"2024-04-24","arxiv_id":"2404.16164","n_code_links":0,"syntology":null},{"paper":"/paper/aligning-llm-agents-by-learning-latent","slug":"aligning-llm-agents-by-learning-latent","title":"Aligning LLM Agents by Learning Latent Preference from User Edits","date":"2024-04-23","arxiv_id":"2404.15269","n_code_links":1,"syntology":null},{"paper":null,"slug":"ct-agent-clinical-trial-multi-agent-with","title":"ClinicalAgent: Clinical Trial Multi-Agent System with Large Language Model-based Reasoning","date":"2024-04-23","arxiv_id":"2404.14777","n_code_links":0,"syntology":null},{"paper":null,"slug":"designprobe-a-graphic-design-benchmark-for","title":"DesignProbe: A Graphic Design Benchmark for Multimodal Large Language Models","date":"2024-04-23","arxiv_id":"2404.14801","n_code_links":0,"syntology":null},{"paper":null,"slug":"prism-patient-records-interpretation-for","title":"PRISM: Patient Records Interpretation for Semantic Clinical Trial Matching using Large Language Models","date":"2024-04-23","arxiv_id":"2404.15549","n_code_links":0,"syntology":null},{"paper":null,"slug":"science-written-by-generative-ai-is-perceived","title":"From Complexity to Clarity: How AI Enhances Perceptions of Scientists and the Public's Understanding of Science","date":"2024-04-23","arxiv_id":"2405.00706","n_code_links":0,"syntology":null},{"paper":"/paper/towards-systematic-evaluation-of-logical","slug":"towards-systematic-evaluation-of-logical","title":"LogicBench: Towards Systematic Evaluation of Logical Reasoning Ability of Large Language Models","date":"2024-04-23","arxiv_id":"2404.15522","n_code_links":1,"syntology":null},{"paper":"/paper/how-well-can-llms-echo-us-evaluating-ai","slug":"how-well-can-llms-echo-us-evaluating-ai","title":"How Well Can LLMs Echo Us? Evaluating AI Chatbots' Role-Play Ability with ECHO","date":"2024-04-22","arxiv_id":"2404.13957","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":0,"n_instrument":6,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cuhk-arise/echo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"information-re-organization-improves","title":"Information Re-Organization Improves Reasoning in Large Language Models","date":"2024-04-22","arxiv_id":"2404.13985","n_code_links":0,"syntology":null},{"paper":null,"slug":"navigating-the-path-of-writing-outline-guided","title":"Navigating the Path of Writing: Outline-guided Text Generation with Large Language Models","date":"2024-04-22","arxiv_id":"2404.13919","n_code_links":0,"syntology":null},{"paper":"/paper/llms-in-web-development-evaluating-llm","slug":"llms-in-web-development-evaluating-llm","title":"LLMs in Web Development: Evaluating LLM-Generated PHP Code Unveiling Vulnerabilities and Limitations","date":"2024-04-21","arxiv_id":"2404.14459","n_code_links":1,"syntology":null},{"paper":"/paper/svgeditbench-a-benchmark-dataset-for","slug":"svgeditbench-a-benchmark-dataset-for","title":"SVGEditBench: A Benchmark Dataset for Quantitative Assessment of LLM's SVG Editing Capabilities","date":"2024-04-21","arxiv_id":"2404.13710","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mti-lab/svgeditbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/beyond-accuracy-investigating-error-types-in","slug":"beyond-accuracy-investigating-error-types-in","title":"Beyond Accuracy: Investigating Error Types in GPT-4 Responses to USMLE Questions","date":"2024-04-20","arxiv_id":"2404.13307","n_code_links":1,"syntology":null},{"paper":"/paper/large-language-models-as-test-case-generators","slug":"large-language-models-as-test-case-generators","title":"Large Language Models as Test Case Generators: Performance Evaluation and Enhancement","date":"2024-04-20","arxiv_id":"2404.13340","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/cross-cultural-inspiration-detection-and","slug":"cross-cultural-inspiration-detection-and","title":"Cross-cultural Inspiration Detection and Analysis in Real and LLM-generated Social Media Data","date":"2024-04-19","arxiv_id":"2404.12933","n_code_links":1,"syntology":null},{"paper":"/paper/cyberseceval-2-a-wide-ranging-cybersecurity","slug":"cyberseceval-2-a-wide-ranging-cybersecurity","title":"CyberSecEval 2: A Wide-Ranging Cybersecurity Evaluation Suite for Large Language Models","date":"2024-04-19","arxiv_id":"2404.13161","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["facebookresearch/purplellama"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/dubo-sql-diverse-retrieval-augmented","slug":"dubo-sql-diverse-retrieval-augmented","title":"Dubo-SQL: Diverse Retrieval-Augmented Generation and Fine Tuning for Text-to-SQL","date":"2024-04-19","arxiv_id":"2404.12560","n_code_links":1,"syntology":null},{"paper":"/paper/advisorqa-towards-helpful-and-harmless-advice","slug":"advisorqa-towards-helpful-and-harmless-advice","title":"AdvisorQA: Towards Helpful and Harmless Advice-seeking Question Answering with Collective Intelligence","date":"2024-04-18","arxiv_id":"2404.11826","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["minbeomkim/advisorqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"bird-a-trustworthy-bayesian-inference","title":"BIRD: A Trustworthy Bayesian Inference Framework for Large Language Models","date":"2024-04-18","arxiv_id":"2404.12494","n_code_links":0,"syntology":null},{"paper":"/paper/caus-a-dataset-for-question-generation-based","slug":"caus-a-dataset-for-question-generation-based","title":"CAUS: A Dataset for Question Generation based on Human Cognition Leveraging Large Language Models","date":"2024-04-18","arxiv_id":"2404.11835","n_code_links":1,"syntology":null},{"paper":null,"slug":"concept-induction-using-llms-a-user","title":"Concept Induction using LLMs: a user experiment for assessment","date":"2024-04-18","arxiv_id":"2404.11875","n_code_links":0,"syntology":null},{"paper":"/paper/openbezoar-small-cost-effective-and-open","slug":"openbezoar-small-cost-effective-and-open","title":"OpenBezoar: Small, Cost-Effective and Open Models Trained on Mixes of Instruction Data","date":"2024-04-18","arxiv_id":"2404.12195","n_code_links":1,"syntology":null},{"paper":"/paper/uncovering-safety-risks-in-open-source-llms","slug":"uncovering-safety-risks-in-open-source-llms","title":"Uncovering Safety Risks of Large Language Models through Concept Activation Vector","date":"2024-04-18","arxiv_id":"2404.12038","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":3,"n_instrument":1,"unverified":4,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["sproutnan/ai-safety_scav"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["found_in_text","official"]}}}],"record_sha256":"9a1ba74c73dc5ff09387356bd4343d1cc4caaf9c28c20ba398608719d4a5a143","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}