{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/19","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":19,"pages_in_order":29,"rows_per_page":100,"rows":[1801,1900],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/18","next":"/method/gpt-4/papers/20","papers":[{"paper":null,"slug":"llms-among-us-generative-ai-participating-in","title":"LLMs Among Us: Generative AI Participating in Digital Discourse","date":"2024-02-08","arxiv_id":"2402.07940","n_code_links":0,"syntology":null},{"paper":"/paper/noise-contrastive-alignment-of-language","slug":"noise-contrastive-alignment-of-language","title":"Noise Contrastive Alignment of Language Models with Explicit Rewards","date":"2024-02-08","arxiv_id":"2402.05369","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thu-ml/noise-contrastive-alignment"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"self-alignment-of-large-language-models-via","title":"Self-Alignment of Large Language Models via Monopolylogue-based Social Scene Simulation","date":"2024-02-08","arxiv_id":"2402.05699","n_code_links":0,"syntology":null},{"paper":null,"slug":"timearena-shaping-efficient-multitasking","title":"TimeArena: Shaping Efficient Multitasking Language Agents in a Time-Aware Simulation","date":"2024-02-08","arxiv_id":"2402.05733","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-chain-of-thought-reasoning-guided","title":"Zero-Shot Chain-of-Thought Reasoning Guided by Evolutionary Algorithms in Large Language Models","date":"2024-02-08","arxiv_id":"2402.05376","n_code_links":0,"syntology":null},{"paper":"/paper/can-large-language-model-agents-simulate","slug":"can-large-language-model-agents-simulate","title":"Can Large Language Model Agents Simulate Human Trust Behavior?","date":"2024-02-07","arxiv_id":"2402.04559","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["camel-ai/agent-trust"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"chatbots-in-knowledge-intensive-contexts","title":"Conversational Assistants in Knowledge-Intensive Contexts: An Evaluation of LLM- versus Intent-based Systems","date":"2024-02-07","arxiv_id":"2402.04955","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-cross-domain-low-resource-text","title":"Improving Cross-Domain Low-Resource Text Generation through LLM Post-Editing: A Programmer-Interpreter Approach","date":"2024-02-07","arxiv_id":"2402.04609","n_code_links":0,"syntology":null},{"paper":"/paper/long-is-more-for-alignment-a-simple-but-tough","slug":"long-is-more-for-alignment-a-simple-but-tough","title":"Long Is More for Alignment: A Simple but Tough-to-Beat Baseline for Instruction Fine-Tuning","date":"2024-02-07","arxiv_id":"2402.04833","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tml-epfl/long-is-more-for-alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"navigating-the-knowledge-sea-planet-scale","title":"Navigating the Knowledge Sea: Planet-scale answer retrieval using LLMs","date":"2024-02-07","arxiv_id":"2402.05318","n_code_links":0,"syntology":null},{"paper":"/paper/opening-the-ai-black-box-program-synthesis","slug":"opening-the-ai-black-box-program-synthesis","title":"Opening the AI black box: program synthesis via mechanistic interpretability","date":"2024-02-07","arxiv_id":"2402.05110","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ejmichaud/neural-verification"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/transllama-llm-based-simultaneous-translation","slug":"transllama-llm-based-simultaneous-translation","title":"TransLLaMa: LLM-based Simultaneous Translation System","date":"2024-02-07","arxiv_id":"2402.04636","n_code_links":1,"syntology":null},{"paper":"/paper/anytool-self-reflective-hierarchical-agents","slug":"anytool-self-reflective-hierarchical-agents","title":"AnyTool: Self-Reflective, Hierarchical Agents for Large-Scale API Calls","date":"2024-02-06","arxiv_id":"2402.04253","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["dyabel/anytool"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"behind-the-screen-investigating-chatgpt-s","title":"Behind the Screen: Investigating ChatGPT's Dark Personality Traits and Conspiracy Beliefs","date":"2024-02-06","arxiv_id":"2402.04110","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-abstraction-in-humans-and-large","title":"Comparing Abstraction in Humans and Large Language Models Using Multimodal Serial Reproduction","date":"2024-02-06","arxiv_id":"2402.03618","n_code_links":0,"syntology":null},{"paper":"/paper/identifying-reasons-for-contraceptive","slug":"identifying-reasons-for-contraceptive","title":"Identifying Reasons for Contraceptive Switching from Real-World Data Using Large Language Models","date":"2024-02-06","arxiv_id":"2402.03597","n_code_links":1,"syntology":null},{"paper":null,"slug":"iterative-prompt-refinement-for-radiation","title":"Iterative Prompt Refinement for Radiation Oncology Symptom Extraction Using Teacher-Student Large Language Models","date":"2024-02-06","arxiv_id":"2402.04075","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-as-moocs-graders","title":"Large Language Models As MOOCs Graders","date":"2024-02-06","arxiv_id":"2402.03776","n_code_links":0,"syntology":null},{"paper":null,"slug":"leak-cheat-repeat-data-contamination-and","title":"Leak, Cheat, Repeat: Data Contamination and Evaluation Malpractices in Closed-Source LLMs","date":"2024-02-06","arxiv_id":"2402.03927","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-agents-can-autonomously-hack-websites","title":"LLM Agents can Autonomously Hack Websites","date":"2024-02-06","arxiv_id":"2402.06664","n_code_links":0,"syntology":null},{"paper":null,"slug":"minds-versus-machines-rethinking-entailment","title":"Are Machines Better at Complex Reasoning? Unveiling Human-Machine Inference Gaps in Entailment Verification","date":"2024-02-06","arxiv_id":"2402.03686","n_code_links":0,"syntology":null},{"paper":null,"slug":"professional-agents-evolving-large-language","title":"Professional Agents -- Evolving Large Language Models into Autonomous Experts with Human-Level Competencies","date":"2024-02-06","arxiv_id":"2402.03628","n_code_links":0,"syntology":null},{"paper":"/paper/self-discover-large-language-models-self","slug":"self-discover-large-language-models-self","title":"Self-Discover: Large Language Models Self-Compose Reasoning Structures","date":"2024-02-06","arxiv_id":"2402.03620","n_code_links":3,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":null}},{"paper":"/paper/conversation-reconstruction-attack-against","slug":"conversation-reconstruction-attack-against","title":"Reconstruct Your Previous Conversations! Comprehensively Investigating Privacy Leakage Risks in Conversations with GPT Models","date":"2024-02-05","arxiv_id":"2402.02987","n_code_links":1,"syntology":null},{"paper":"/paper/deepseekmath-pushing-the-limits-of","slug":"deepseekmath-pushing-the-limits-of","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","date":"2024-02-05","arxiv_id":"2402.03300","n_code_links":5,"syntology":{"ran":14,"of":24,"n_ran_checked":10,"n_instrument":4,"unverified":10,"pointer_only":3,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 4 where Syntology's instrument failed) · 10 unverified","official":{"repos":["deepseek-ai/deepseek-math"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/graph-enhanced-large-language-models-in","slug":"graph-enhanced-large-language-models-in","title":"Graph-enhanced Large Language Models in Asynchronous Plan Reasoning","date":"2024-02-05","arxiv_id":"2402.02805","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":0,"n_instrument":6,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","official":{"repos":["fangru-lin/graph-llm-asynchow-plan"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"harnessing-pubmed-user-query-logs-for-post","title":"Harnessing PubMed User Query Logs for Post Hoc Explanations of Recommended Similar Articles","date":"2024-02-05","arxiv_id":"2402.03484","n_code_links":0,"syntology":null},{"paper":"/paper/is-mamba-capable-of-in-context-learning","slug":"is-mamba-capable-of-in-context-learning","title":"Is Mamba Capable of In-Context Learning?","date":"2024-02-05","arxiv_id":"2402.03170","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["automl/is_mamba_capable_of_icl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/swag-storytelling-with-action-guidance","slug":"swag-storytelling-with-action-guidance","title":"SWAG: Storytelling With Action Guidance","date":"2024-02-05","arxiv_id":"2402.03483","n_code_links":1,"syntology":null},{"paper":null,"slug":"aligner-achieving-efficient-alignment-through","title":"Aligner: Efficient Alignment by Learning to Correct","date":"2024-02-04","arxiv_id":"2402.02416","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-large-language-models-in-analysing","title":"Evaluating Large Language Models in Analysing Classroom Dialogue","date":"2024-02-04","arxiv_id":"2402.02380","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-assessment-of-tutoring-practices","title":"Improving Assessment of Tutoring Practices using Retrieval-Augmented Generation","date":"2024-02-04","arxiv_id":"2402.14594","n_code_links":0,"syntology":null},{"paper":null,"slug":"betterv-controlled-verilog-generation-with","title":"BetterV: Controlled Verilog Generation with Discriminative Guidance","date":"2024-02-03","arxiv_id":"2402.03375","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-moral-judgment-and-reasoning-capability-of","title":"Do Moral Judgment and Reasoning Capability of LLMs Change with Language? A Study using the Multilingual Defining Issues Test","date":"2024-02-03","arxiv_id":"2402.02135","n_code_links":0,"syntology":null},{"paper":"/paper/effibench-benchmarking-the-efficiency-of","slug":"effibench-benchmarking-the-efficiency-of","title":"EffiBench: Benchmarking the Efficiency of Automatically Generated Code","date":"2024-02-03","arxiv_id":"2402.02037","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":1,"n_instrument":2,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["huangd1999/EffiBench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-well-do-llms-cite-relevant-medical","slug":"how-well-do-llms-cite-relevant-medical","title":"How well do LLMs cite relevant medical references? An evaluation framework and analyses","date":"2024-02-03","arxiv_id":"2402.02008","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-can-generative-ai-enhance-the-well-being","title":"How Can Generative AI Enhance the Well-being of Blind?","date":"2024-02-02","arxiv_id":"2402.07919","n_code_links":0,"syntology":null},{"paper":"/paper/integrating-large-language-models-in-causal","slug":"integrating-large-language-models-in-causal","title":"Integrating Large Language Models in Causal Discovery: A Statistical Causal Approach","date":"2024-02-02","arxiv_id":"2402.01454","n_code_links":2,"syntology":null},{"paper":"/paper/travelplanner-a-benchmark-for-real-world","slug":"travelplanner-a-benchmark-for-real-world","title":"TravelPlanner: A Benchmark for Real-World Planning with Language Agents","date":"2024-02-02","arxiv_id":"2402.01622","n_code_links":2,"syntology":{"ran":17,"of":17,"n_ran_checked":13,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["OSU-NLP-Group/TravelPlanner"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"generation-distillation-and-evaluation-of","title":"Generation, Distillation and Evaluation of Motivational Interviewing-Style Reflections with a Foundational Language Model","date":"2024-02-01","arxiv_id":"2402.01051","n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-multi-label-classification-of-2","title":"Hierarchical Multi-Label Classification of Online Vaccine Concerns","date":"2024-02-01","arxiv_id":"2402.01783","n_code_links":0,"syntology":null},{"paper":null,"slug":"ocassionally-secure-a-comparative-analysis-of","title":"Ocassionally Secure: A Comparative Analysis of Code Generation Assistants","date":"2024-02-01","arxiv_id":"2402.00689","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-psychology-of-gpt-4-moderately-anxious","title":"On the Psychology of GPT-4: Moderately anxious, slightly masculine, honest, and humble","date":"2024-02-01","arxiv_id":"2402.01777","n_code_links":0,"syntology":null},{"paper":null,"slug":"code-aware-prompting-a-study-of-coverage","title":"Code-Aware Prompting: A study of Coverage Guided Test Generation in Regression Setting using LLM","date":"2024-01-31","arxiv_id":"2402.00097","n_code_links":0,"syntology":null},{"paper":null,"slug":"global-liar-factuality-of-llms-over-time-and","title":"Global-Liar: Factuality of LLMs over Time and Geographic Regions","date":"2024-01-31","arxiv_id":"2401.17839","n_code_links":0,"syntology":null},{"paper":"/paper/llm-voting-human-choices-and-ai-collective","slug":"llm-voting-human-choices-and-ai-collective","title":"LLM Voting: Human Choices and AI Collective Decision Making","date":"2024-01-31","arxiv_id":"2402.01766","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["ethz-coss/LLM_voting"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/raptor-recursive-abstractive-processing-for","slug":"raptor-recursive-abstractive-processing-for","title":"RAPTOR: Recursive Abstractive Processing for Tree-Organized Retrieval","date":"2024-01-31","arxiv_id":"2401.18059","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["parthsarthi03/RAPTOR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/scape-searching-conceptual-architecture","slug":"scape-searching-conceptual-architecture","title":"SCAPE: Searching Conceptual Architecture Prompts using Evolution","date":"2024-01-31","arxiv_id":"2402.00089","n_code_links":1,"syntology":null},{"paper":null,"slug":"scavenging-hyena-distilling-transformers-into","title":"Scavenging Hyena: Distilling Transformers into Long Convolution Models","date":"2024-01-31","arxiv_id":"2401.17574","n_code_links":0,"syntology":null},{"paper":null,"slug":"supporting-anticipatory-governance-using-llms","title":"Evaluating the Capabilities of LLMs for Supporting Anticipatory Impact Assessment","date":"2024-01-31","arxiv_id":"2401.18028","n_code_links":0,"syntology":null},{"paper":"/paper/wsc-enhancing-the-winograd-schema-challenge","slug":"wsc-enhancing-the-winograd-schema-challenge","title":"WSC+: Enhancing The Winograd Schema Challenge Using Tree-of-Experts","date":"2024-01-31","arxiv_id":"2401.17703","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-cross-language-investigation-into-jailbreak","title":"A Cross-Language Investigation into Jailbreak Attacks in Large Language Models","date":"2024-01-30","arxiv_id":"2401.16765","n_code_links":0,"syntology":null},{"paper":"/paper/conditional-and-modal-reasoning-in-large","slug":"conditional-and-modal-reasoning-in-large","title":"Conditional and Modal Reasoning in Large Language Models","date":"2024-01-30","arxiv_id":"2401.17169","n_code_links":1,"syntology":null},{"paper":"/paper/mt-eval-a-multi-turn-capabilities-evaluation","slug":"mt-eval-a-multi-turn-capabilities-evaluation","title":"MT-Eval: A Multi-Turn Capabilities Evaluation Benchmark for Large Language Models","date":"2024-01-30","arxiv_id":"2401.16745","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["kwanwaichung/mt-eval"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"performance-assessment-of-chatgpt-vs-bard-in","title":"Performance Assessment of ChatGPT vs Bard in Detecting Alzheimer's Dementia","date":"2024-01-30","arxiv_id":"2402.01751","n_code_links":0,"syntology":null},{"paper":"/paper/robust-prompt-optimization-for-defending","slug":"robust-prompt-optimization-for-defending","title":"Robust Prompt Optimization for Defending Language Models Against Jailbreaking Attacks","date":"2024-01-30","arxiv_id":"2401.17263","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lapisrocks/rpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/synthetic-dialogue-dataset-generation-using","slug":"synthetic-dialogue-dataset-generation-using","title":"Synthetic Dialogue Dataset Generation using LLM Agents","date":"2024-01-30","arxiv_id":"2401.17461","n_code_links":1,"syntology":null},{"paper":null,"slug":"weaver-foundation-models-for-creative-writing","title":"Weaver: Foundation Models for Creative Writing","date":"2024-01-30","arxiv_id":"2401.17268","n_code_links":0,"syntology":null},{"paper":"/paper/3dg-a-framework-for-using-generative-ai-for","slug":"3dg-a-framework-for-using-generative-ai-for","title":"3DG: A Framework for Using Generative AI for Handling Sparse Learner Performance Data From Intelligent Tutoring Systems","date":"2024-01-29","arxiv_id":"2402.01746","n_code_links":1,"syntology":null},{"paper":null,"slug":"development-and-testing-of-a-novel-large","title":"Development and Testing of a Novel Large Language Model-Based Clinical Decision Support Systems for Medication Safety in 12 Clinical Specialties","date":"2024-01-29","arxiv_id":"2402.01741","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-professional-radiologists","title":"Leveraging Professional Radiologists' Expertise to Enhance LLMs' Evaluation for Radiology Reports","date":"2024-01-29","arxiv_id":"2401.16578","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm4vuln-a-unified-evaluation-framework-for","title":"LLM4Vuln: A Unified Evaluation Framework for Decoupling and Enhancing LLMs' Vulnerability Reasoning","date":"2024-01-29","arxiv_id":"2401.16185","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt4vis-prompting-large-language-models","title":"Prompt4Vis: Prompting Large Language Models with Example Mining and Schema Filtering for Tabular Data Visualization","date":"2024-01-29","arxiv_id":"2402.07909","n_code_links":0,"syntology":null},{"paper":null,"slug":"response-generation-for-cognitive-behavioral","title":"Response Generation for Cognitive Behavioral Therapy with Large Language Models: Comparative Study with Socratic Questioning","date":"2024-01-29","arxiv_id":"2401.15966","n_code_links":0,"syntology":null},{"paper":null,"slug":"security-code-review-by-llms-a-deep-dive-into","title":"An Insight into Security Code Review with LLMs: Capabilities, Obstacles, and Influential Factors","date":"2024-01-29","arxiv_id":"2401.16310","n_code_links":0,"syntology":null},{"paper":null,"slug":"identifying-and-improving-disability-bias-in","title":"Identifying and Improving Disability Bias in GPT-Based Resume Screening","date":"2024-01-28","arxiv_id":"2402.01732","n_code_links":0,"syntology":null},{"paper":null,"slug":"pre-a-peer-review-based-large-language-model","title":"PRE: A Peer Review Based Large Language Model Evaluator","date":"2024-01-28","arxiv_id":"2401.15641","n_code_links":0,"syntology":null},{"paper":null,"slug":"dataframe-qa-a-universal-llm-framework-on","title":"DataFrame QA: A Universal LLM Framework on DataFrame Question Answering Without Data Exposure","date":"2024-01-27","arxiv_id":"2401.15463","n_code_links":0,"syntology":null},{"paper":"/paper/improving-medical-reasoning-through-retrieval","slug":"improving-medical-reasoning-through-retrieval","title":"Improving Medical Reasoning through Retrieval and Self-Reflection with Retrieval-Augmented Large Language Models","date":"2024-01-27","arxiv_id":"2401.15269","n_code_links":1,"syntology":null},{"paper":"/paper/multihop-rag-benchmarking-retrieval-augmented","slug":"multihop-rag-benchmarking-retrieval-augmented","title":"MultiHop-RAG: Benchmarking Retrieval-Augmented Generation for Multi-Hop Queries","date":"2024-01-27","arxiv_id":"2401.15391","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yixuantt/MultiHop-RAG"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"prompting-diverse-ideas-increasing-ai-idea","title":"Prompting Diverse Ideas: Increasing AI Idea Variance","date":"2024-01-27","arxiv_id":"2402.01727","n_code_links":0,"syntology":null},{"paper":"/paper/chemdfm-dialogue-foundation-model-for","slug":"chemdfm-dialogue-foundation-model-for","title":"ChemDFM: A Large Language Foundation Model for Chemistry","date":"2024-01-26","arxiv_id":"2401.14818","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"enhancing-diagnostic-accuracy-through-multi","title":"Enhancing Diagnostic Accuracy through Multi-Agent Conversations: Using Large Language Models to Mitigate Cognitive Bias","date":"2024-01-26","arxiv_id":"2401.14589","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-llm-chatbots-for-osint-based","title":"Evaluation of LLM Chatbots for OSINT-based Cyber Threat Awareness","date":"2024-01-26","arxiv_id":"2401.15127","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-gpt-4-to-gemini-and-beyond-assessing-the","title":"From GPT-4 to Gemini and Beyond: Assessing the Landscape of MLLMs on Generalizability, Trustworthiness and Causality through Four Modalities","date":"2024-01-26","arxiv_id":"2401.15071","n_code_links":0,"syntology":null},{"paper":"/paper/health-text-simplification-an-annotated","slug":"health-text-simplification-an-annotated","title":"Health Text Simplification: An Annotated Corpus for Digestive Cancer Education and Novel Strategies for Reinforcement Learning","date":"2024-01-26","arxiv_id":"2401.15043","n_code_links":1,"syntology":null},{"paper":null,"slug":"scalable-qualitative-coding-with-llms-chain","title":"Scalable Qualitative Coding with LLMs: Chain-of-Thought Reasoning Matches Human Performance in Some Hermeneutic Tasks","date":"2024-01-26","arxiv_id":"2401.15170","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-zero-shot-inference","title":"A comparative study of zero-shot inference with large language models and supervised modeling in breast cancer pathology classification","date":"2024-01-25","arxiv_id":"2401.13887","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigate-consolidate-exploit-a-general","title":"Investigate-Consolidate-Exploit: A General Strategy for Inter-Task Agent Self-Evolution","date":"2024-01-25","arxiv_id":"2401.13996","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-on-fhir-demystifying-health-records","title":"LLM on FHIR -- Demystifying Health Records","date":"2024-01-25","arxiv_id":"2402.01711","n_code_links":0,"syntology":null},{"paper":"/paper/prompting-large-language-models-for-zero-shot-1","slug":"prompting-large-language-models-for-zero-shot-1","title":"Prompting Large Language Models for Zero-Shot Clinical Prediction with Structured Longitudinal Electronic Health Record Data","date":"2024-01-25","arxiv_id":"2402.01713","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yhzhu99/llm4healthcare"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"unmasking-and-quantifying-racial-bias-of","title":"Unmasking and Quantifying Racial Bias of Large Language Models in Medical Report Generation","date":"2024-01-25","arxiv_id":"2401.13867","n_code_links":0,"syntology":null},{"paper":"/paper/webvoyager-building-an-end-to-end-web-agent","slug":"webvoyager-building-an-end-to-end-web-agent","title":"WebVoyager: Building an End-to-End Web Agent with Large Multimodal Models","date":"2024-01-25","arxiv_id":"2401.13919","n_code_links":2,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["minorjerry/webvoyager"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"zero-shot-sequential-neuro-symbolic-reasoning","title":"Zero-shot Sequential Neuro-symbolic Reasoning for Automatically Generating Architecture Schematic Designs","date":"2024-01-25","arxiv_id":"2402.00052","n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-root-causing-of-cloud-incidents","title":"Automated Root Causing of Cloud Incidents using In-Context Learning with GPT-4","date":"2024-01-24","arxiv_id":"2401.13810","n_code_links":0,"syntology":null},{"paper":"/paper/clue-guided-path-exploration-an-efficient","slug":"clue-guided-path-exploration-an-efficient","title":"Fine-Grained Stateful Knowledge Exploration: A Novel Paradigm for Integrating Knowledge Graphs with Large Language Models","date":"2024-01-24","arxiv_id":"2401.13444","n_code_links":1,"syntology":null},{"paper":"/paper/contextual-evaluating-context-sensitive-text","slug":"contextual-evaluating-context-sensitive-text","title":"ConTextual: Evaluating Context-Sensitive Text-Rich Visual Reasoning in Large Multimodal Models","date":"2024-01-24","arxiv_id":"2401.13311","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rohan598/contextual"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluation-of-general-large-language-models","title":"Evaluation of General Large Language Models in Contextually Assessing Semantic Concepts Extracted from Adult Critical Care Electronic Health Record Notes","date":"2024-01-24","arxiv_id":"2401.13588","n_code_links":0,"syntology":null},{"paper":"/paper/how-good-is-chatgpt-at-face-biometrics-a","slug":"how-good-is-chatgpt-at-face-biometrics-a","title":"How Good is ChatGPT at Face Biometrics? A First Look into Recognition, Soft Biometrics, and Explainability","date":"2024-01-24","arxiv_id":"2401.13641","n_code_links":1,"syntology":null},{"paper":null,"slug":"research-about-the-ability-of-llm-in-the","title":"Research about the Ability of LLM in the Tamper-Detection Area","date":"2024-01-24","arxiv_id":"2401.13504","n_code_links":0,"syntology":null},{"paper":null,"slug":"tat-llm-a-specialized-language-model-for","title":"TAT-LLM: A Specialized Language Model for Discrete Reasoning over Tabular and Textual Data","date":"2024-01-24","arxiv_id":"2401.13223","n_code_links":0,"syntology":null},{"paper":"/paper/args-alignment-as-reward-guided-search","slug":"args-alignment-as-reward-guided-search","title":"ARGS: Alignment as Reward-Guided Search","date":"2024-01-23","arxiv_id":"2402.01694","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["deeplearning-wisc/args"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"kam-cot-knowledge-augmented-multimodal-chain","title":"KAM-CoT: Knowledge Augmented Multimodal Chain-of-Thoughts Reasoning","date":"2024-01-23","arxiv_id":"2401.12863","n_code_links":0,"syntology":null},{"paper":"/paper/meta-prompting-enhancing-language-models-with","slug":"meta-prompting-enhancing-language-models-with","title":"Meta-Prompting: Enhancing Language Models with Task-Agnostic Scaffolding","date":"2024-01-23","arxiv_id":"2401.12954","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["suzgunmirac/meta-prompting"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"quality-of-answers-of-generative-large","title":"Quality of Answers of Generative Large Language Models vs Peer Patients for Interpreting Lab Test Results for Lay Patients: Evaluation Study","date":"2024-01-23","arxiv_id":"2402.01693","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-large-language-models-for","title":"Investigating Large Language Models for Financial Causality Detection in Multilingual Setup","date":"2024-01-22","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"revolutionizing-finance-with-llms-an-overview","title":"Revolutionizing Finance with LLMs: An Overview of Applications and Insights","date":"2024-01-22","arxiv_id":"2401.11641","n_code_links":0,"syntology":null},{"paper":"/paper/speak-it-out-solving-symbol-related-problems","slug":"speak-it-out-solving-symbol-related-problems","title":"Speak It Out: Solving Symbol-Related Problems with Symbol-to-Language Conversion for Language Models","date":"2024-01-22","arxiv_id":"2401.11725","n_code_links":1,"syntology":null},{"paper":"/paper/superclue-math6-graded-multi-step-math","slug":"superclue-math6-graded-multi-step-math","title":"SuperCLUE-Math6: Graded Multi-Step Math Reasoning Benchmark for LLMs in Chinese","date":"2024-01-22","arxiv_id":"2401.11819","n_code_links":1,"syntology":null},{"paper":"/paper/prolex-a-benchmark-for-language-proficiency","slug":"prolex-a-benchmark-for-language-proficiency","title":"ProLex: A Benchmark for Language Proficiency-oriented Lexical Substitution","date":"2024-01-21","arxiv_id":"2401.11356","n_code_links":1,"syntology":null}],"record_sha256":"8109c21528d2ce84199de59ff4b31af9c444a9163f9dd295d88c50b83c5d374c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}