{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/24","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":24,"pages_in_order":29,"rows_per_page":100,"rows":[2301,2400],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/23","next":"/method/gpt-4/papers/25","papers":[{"paper":null,"slug":"diversity-of-thought-improves-reasoning","title":"Diversity of Thought Improves Reasoning Abilities of LLMs","date":"2023-10-11","arxiv_id":"2310.07088","n_code_links":0,"syntology":null},{"paper":null,"slug":"ethical-reasoning-over-moral-alignment-a-case","title":"Ethical Reasoning over Moral Alignment: A Case and Framework for In-Context Ethical Policies in LLMs","date":"2023-10-11","arxiv_id":"2310.07251","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-landscape-of-large-language","title":"Do Large Language Models have Shared Weaknesses in Medical Question Answering?","date":"2023-10-11","arxiv_id":"2310.07225","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-are-zero-shot-time-1","slug":"large-language-models-are-zero-shot-time-1","title":"Large Language Models Are Zero-Shot Time Series Forecasters","date":"2023-10-11","arxiv_id":"2310.07820","n_code_links":2,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ngruver/llmtime"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"towards-foundation-models-for-learning-on","title":"From Supervised to Generative: A Novel Paradigm for Tabular Deep Learning with Large Language Models","date":"2023-10-11","arxiv_id":"2310.07338","n_code_links":0,"syntology":null},{"paper":null,"slug":"wigenai-the-symphony-of-wireless-and","title":"Diffusion Models for Wireless Communications","date":"2023-10-11","arxiv_id":"2310.07312","n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-clinical-coding-using-off-the-shelf","title":"Automated clinical coding using off-the-shelf large language models","date":"2023-10-10","arxiv_id":"2310.06552","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-and-evaluating-tests-for-k-12","title":"Generating and Evaluating Tests for K-12 Students with Language Model Simulations: A Case Study on Sentence Reading Efficiency","date":"2023-10-10","arxiv_id":"2310.06837","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-as-an-agronomist-assistant-answering","title":"GPT-4 as an Agronomist Assistant? Answering Agriculture Exams Using Large Language Models","date":"2023-10-10","arxiv_id":"2310.06225","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-for-propaganda","slug":"large-language-models-for-propaganda","title":"Large Language Models for Propaganda Detection","date":"2023-10-10","arxiv_id":"2310.06422","n_code_links":2,"syntology":null},{"paper":null,"slug":"llms-as-potential-brainstorming-partners-for","title":"LLMs as Potential Brainstorming Partners for Math and Science Problems","date":"2023-10-10","arxiv_id":"2310.10677","n_code_links":0,"syntology":null},{"paper":"/paper/multilingual-jailbreak-challenges-in-large","slug":"multilingual-jailbreak-challenges-in-large","title":"Multilingual Jailbreak Challenges in Large Language Models","date":"2023-10-10","arxiv_id":"2310.06474","n_code_links":1,"syntology":null},{"paper":null,"slug":"newton-are-large-language-models-capable-of","title":"NEWTON: Are Large Language Models Capable of Physical Reasoning?","date":"2023-10-10","arxiv_id":"2310.07018","n_code_links":0,"syntology":null},{"paper":"/paper/swe-bench-can-language-models-resolve-real","slug":"swe-bench-can-language-models-resolve-real","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","date":"2023-10-10","arxiv_id":"2310.06770","n_code_links":8,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/what-if-the-tv-was-off-examining","slug":"what-if-the-tv-was-off-examining","title":"What If the TV Was Off? Examining Counterfactual Reasoning Abilities of Multi-modal Language Models","date":"2023-10-10","arxiv_id":"2310.06627","n_code_links":1,"syntology":null},{"paper":"/paper/fireact-toward-language-agent-fine-tuning","slug":"fireact-toward-language-agent-fine-tuning","title":"FireAct: Toward Language Agent Fine-tuning","date":"2023-10-09","arxiv_id":"2310.05915","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrating-graphs-with-large-language-models","title":"Integrating Graphs with Large Language Models: Methods and Prospects","date":"2023-10-09","arxiv_id":"2310.05499","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrating-stock-features-and-global","title":"Integrating Stock Features and Global Information via Large Language Models for Enhanced Stock Return Prediction","date":"2023-10-09","arxiv_id":"2310.05627","n_code_links":0,"syntology":null},{"paper":"/paper/put-your-money-where-your-mouth-is-evaluating","slug":"put-your-money-where-your-mouth-is-evaluating","title":"Put Your Money Where Your Mouth Is: Evaluating Strategic Planning and Execution of LLM Agents in an Auction Arena","date":"2023-10-09","arxiv_id":"2310.05746","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jiangjiechen/auction-arena"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/reinforcement-learning-in-the-era-of-llms","slug":"reinforcement-learning-in-the-era-of-llms","title":"Reinforcement Learning in the Era of LLMs: What is Essential? What is needed? An RL Perspective on RLHF, Prompting, and Beyond","date":"2023-10-09","arxiv_id":"2310.06147","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"sc-safety-a-multi-round-open-ended-question","title":"SC-Safety: A Multi-round Open-ended Question Adversarial Safety Benchmark for Large Language Models in Chinese","date":"2023-10-09","arxiv_id":"2310.05818","n_code_links":0,"syntology":null},{"paper":null,"slug":"take-a-step-back-evoking-reasoning-via","title":"Take a Step Back: Evoking Reasoning via Abstraction in Large Language Models","date":"2023-10-09","arxiv_id":"2310.06117","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-importance-of-prompt-tuning-for-automated","title":"The Importance of Prompt Tuning for Automated Neuron Explanations","date":"2023-10-09","arxiv_id":"2310.06200","n_code_links":0,"syntology":null},{"paper":null,"slug":"chatradio-valuer-a-chat-large-language-model","title":"ChatRadio-Valuer: A Chat Large Language Model for Generalizable Radiology Report Generation Based on Multi-institution and Multi-system Data","date":"2023-10-08","arxiv_id":"2310.05242","n_code_links":0,"syntology":null},{"paper":"/paper/harnessing-the-power-of-large-language-models-1","slug":"harnessing-the-power-of-large-language-models-1","title":"Harnessing the Power of Large Language Models for Empathetic Response Generation: Empirical Investigations and Improvements","date":"2023-10-08","arxiv_id":"2310.05140","n_code_links":1,"syntology":{"ran":1,"of":5,"n_ran_checked":1,"n_instrument":0,"unverified":4,"pointer_only":5,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["27182812/LLM4ED"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/llm4vv-developing-llm-driven-testsuite-for","slug":"llm4vv-developing-llm-driven-testsuite-for","title":"LLM4VV: Developing LLM-Driven Testsuite for Compiler Validation","date":"2023-10-08","arxiv_id":"2310.04963","n_code_links":1,"syntology":null},{"paper":"/paper/zero-shot-detection-of-machine-generated","slug":"zero-shot-detection-of-machine-generated","title":"Zero-Shot Detection of Machine-Generated Codes","date":"2023-10-08","arxiv_id":"2310.05103","n_code_links":1,"syntology":null},{"paper":null,"slug":"diffnas-bootstrapping-diffusion-models-by","title":"DiffNAS: Bootstrapping Diffusion Models by Prompting for Better Architectures","date":"2023-10-07","arxiv_id":"2310.04750","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-for-spatial-trajectory","title":"Large Language Models for Spatial Trajectory Patterns Mining","date":"2023-10-07","arxiv_id":"2310.04942","n_code_links":0,"syntology":null},{"paper":"/paper/a-language-agent-approach-to-formal-theorem","slug":"a-language-agent-approach-to-formal-theorem","title":"An In-Context Learning Agent for Formal Theorem-Proving","date":"2023-10-06","arxiv_id":"2310.04353","n_code_links":1,"syntology":null},{"paper":null,"slug":"coding-by-design-gpt-4-empowers-agile-model","title":"Coding by Design: GPT-4 empowers Agile Model Driven Development","date":"2023-10-06","arxiv_id":"2310.04304","n_code_links":0,"syntology":null},{"paper":"/paper/language-agent-tree-search-unifies-reasoning","slug":"language-agent-tree-search-unifies-reasoning","title":"Language Agent Tree Search Unifies Reasoning Acting and Planning in Language Models","date":"2023-10-06","arxiv_id":"2310.04406","n_code_links":2,"syntology":{"ran":9,"of":9,"n_ran_checked":8,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lapisrocks/languageagenttreesearch","andyz245/LanguageAgentTreeSearch"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"adversarial-machine-learning-for-social-good","title":"Adversarial Machine Learning for Social Good: Reframing the Adversary as an Ally","date":"2023-10-05","arxiv_id":"2310.03614","n_code_links":0,"syntology":null},{"paper":"/paper/automating-human-tutor-style-programming","slug":"automating-human-tutor-style-programming","title":"Automating Human Tutor-Style Programming Feedback: Leveraging GPT-4 Tutor Model for Hint Generation and GPT-3.5 Student Model for Hint Validation","date":"2023-10-05","arxiv_id":"2310.03780","n_code_links":2,"syntology":{"ran":10,"of":10,"n_ran_checked":10,"n_instrument":0,"unverified":0,"pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["machine-teaching-group/lak2024_gpt4-hints-gpt3.5val","machine-teaching-group/lak2024_gpt4hints-gpt3.5val"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"benchmarking-a-foundation-llm-on-its-ability","title":"Benchmarking a foundation LLM on its ability to re-label structure names in accordance with the AAPM TG-263 report","date":"2023-10-05","arxiv_id":"2310.03874","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-large-language-models-as-ai","slug":"benchmarking-large-language-models-as-ai","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","date":"2023-10-05","arxiv_id":"2310.03302","n_code_links":2,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["snap-stanford/mlagentbench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-large-language-models-be-good-path","slug":"can-large-language-models-be-good-path","title":"Can Large Language Models be Good Path Planners? A Benchmark and Investigation on Spatial-temporal Reasoning","date":"2023-10-05","arxiv_id":"2310.03249","n_code_links":1,"syntology":{"ran":28,"of":28,"n_ran_checked":28,"n_instrument":0,"unverified":0,"pointer_only":28,"phrase":"28 ran (of which 0 constructed an object rather than computing a result; 28 with no instrument failure: 0 honoured, 0 violated, 28 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mohamedaghzal/llms-as-path-planners"],"state":"official (archive's flag): 28 ran","n_ran":28,"n_constructed":0,"n_ran_no_instrument_failure":28,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-hallucinations-in-chinese-large","slug":"evaluating-hallucinations-in-chinese-large","title":"Evaluating Hallucinations in Chinese Large Language Models","date":"2023-10-05","arxiv_id":"2310.03368","n_code_links":3,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xiami2019/halluqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"heap-hierarchical-policies-for-web-actions","title":"SteP: Stacked LLM Policies for Web Actions","date":"2023-10-05","arxiv_id":"2310.03720","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-personalized-story-evaluation","title":"Learning Personalized Alignment for Evaluating Open-ended Text Generation","date":"2023-10-05","arxiv_id":"2310.03304","n_code_links":0,"syntology":null},{"paper":"/paper/mathcoder-seamless-code-integration-in-llms","slug":"mathcoder-seamless-code-integration-in-llms","title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","date":"2023-10-05","arxiv_id":"2310.03731","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mathllm/mathcoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/procedural-text-mining-with-large-language","slug":"procedural-text-mining-with-large-language","title":"Procedural Text Mining with Large Language Models","date":"2023-10-05","arxiv_id":"2310.03376","n_code_links":1,"syntology":null},{"paper":"/paper/reformulating-domain-adaptation-of-large","slug":"reformulating-domain-adaptation-of-large","title":"Reformulating Domain Adaptation of Large Language Models as Adapt-Retrieve-Revise: A Case Study on Chinese Legal Domain","date":"2023-10-05","arxiv_id":"2310.03328","n_code_links":1,"syntology":null},{"paper":"/paper/can-language-models-employ-the-socratic","slug":"can-language-models-employ-the-socratic","title":"Can Language Models Employ the Socratic Method? Experiments with Code Debugging","date":"2023-10-04","arxiv_id":"2310.03210","n_code_links":1,"syntology":null},{"paper":null,"slug":"citing-large-language-models-create","title":"CITING: Large Language Models Create Curriculum for Instruction Tuning","date":"2023-10-04","arxiv_id":"2310.02527","n_code_links":0,"syntology":null},{"paper":"/paper/dq-lore-dual-queries-with-low-rank","slug":"dq-lore-dual-queries-with-low-rank","title":"DQ-LoRe: Dual Queries with Low Rank Approximation Re-ranking for In-Context Learning","date":"2023-10-04","arxiv_id":"2310.02954","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":3,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ai4fun/dq-lore"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"paper":null,"slug":"gpt-4-as-an-interface-between-researchers-and","title":"GPT-4 as an interface between researchers and computational software: improving usability and reproducibility","date":"2023-10-04","arxiv_id":"2310.11458","n_code_links":0,"syntology":null},{"paper":"/paper/how-far-are-large-language-models-from-agents","slug":"how-far-are-large-language-models-from-agents","title":"How FaR Are Large Language Models From Agents with Theory-of-Mind?","date":"2023-10-04","arxiv_id":"2310.03051","n_code_links":1,"syntology":null},{"paper":"/paper/large-language-model-cascades-with-mixture-of","slug":"large-language-model-cascades-with-mixture-of","title":"Large Language Model Cascades with Mixture of Thoughts Representations for Cost-efficient Reasoning","date":"2023-10-04","arxiv_id":"2310.03094","n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-question-answering-for-unified","slug":"multimodal-question-answering-for-unified","title":"Multimodal Question Answering for Unified Information Extraction","date":"2023-10-04","arxiv_id":"2310.03017","n_code_links":1,"syntology":null},{"paper":null,"slug":"robust-and-interpretable-medical-image","title":"Robust and Interpretable Medical Image Classifiers via Concept Bottleneck Models","date":"2023-10-04","arxiv_id":"2310.03182","n_code_links":0,"syntology":null},{"paper":"/paper/t-3-bench-benchmarking-current-progress-in","slug":"t-3-bench-benchmarking-current-progress-in","title":"T$^3$Bench: Benchmarking Current Progress in Text-to-3D Generation","date":"2023-10-04","arxiv_id":"2310.02977","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["THU-LYJ-Lab/T3Bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"understanding-in-context-learning-in","title":"Understanding In-Context Learning in Transformers and LLMs by Learning to Learn Discrete Functions","date":"2023-10-04","arxiv_id":"2310.03016","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-and-improving-generator","title":"Benchmarking and Improving Generator-Validator Consistency of Language Models","date":"2023-10-03","arxiv_id":"2310.01846","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-gpt-4-replicate-empirical-software","title":"Can GPT-4 Replicate Empirical Software Engineering Research?","date":"2023-10-03","arxiv_id":"2310.01727","n_code_links":0,"syntology":null},{"paper":"/paper/can-large-language-models-provide-useful","slug":"can-large-language-models-provide-useful","title":"Can large language models provide useful feedback on research papers? A large-scale empirical analysis","date":"2023-10-03","arxiv_id":"2310.01783","n_code_links":1,"syntology":null},{"paper":"/paper/contrastive-post-training-large-language","slug":"contrastive-post-training-large-language","title":"Automatic Pair Construction for Contrastive Post-training","date":"2023-10-03","arxiv_id":"2310.02263","n_code_links":1,"syntology":null},{"paper":"/paper/ecoassistant-using-llm-assistant-more","slug":"ecoassistant-using-llm-assistant-more","title":"EcoAssistant: Using LLM Assistant More Affordably and Accurately","date":"2023-10-03","arxiv_id":"2310.03046","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jieyuz2/ecoassistant"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/editing-personality-for-llms","slug":"editing-personality-for-llms","title":"Editing Personality for Large Language Models","date":"2023-10-03","arxiv_id":"2310.02168","n_code_links":1,"syntology":null},{"paper":"/paper/halle-switch-rethinking-and-controlling","slug":"halle-switch-rethinking-and-controlling","title":"HallE-Control: Controlling Object Hallucination in Large Multimodal Models","date":"2023-10-03","arxiv_id":"2310.01779","n_code_links":2,"syntology":{"ran":6,"of":9,"n_ran_checked":3,"n_instrument":3,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["bronyayang/HallE_Switch","bronyayang/halle_control"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/instance-needs-more-care-rewriting-prompts","slug":"instance-needs-more-care-rewriting-prompts","title":"Instances Need More Care: Rewriting Prompts for Instances with LLMs in the Loop Yields Better Zero-Shot Performance","date":"2023-10-03","arxiv_id":"2310.02107","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["salokr/propmted"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"investigating-large-language-models","title":"Investigating Large Language Models' Perception of Emotion Using Appraisal Theory","date":"2023-10-03","arxiv_id":"2310.04450","n_code_links":0,"syntology":null},{"paper":null,"slug":"low-resource-languages-jailbreak-gpt-4","title":"Low-Resource Languages Jailbreak GPT-4","date":"2023-10-03","arxiv_id":"2310.02446","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-from-automatic","title":"Reinforcement Learning from Automatic Feedback for High-Quality Unit Test Generation","date":"2023-10-03","arxiv_id":"2310.02368","n_code_links":0,"syntology":null},{"paper":"/paper/self-taught-optimizer-stop-recursively-self","slug":"self-taught-optimizer-stop-recursively-self","title":"Self-Taught Optimizer (STOP): Recursively Self-Improving Code Generation","date":"2023-10-03","arxiv_id":"2310.02304","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["microsoft/stop"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"what-s-next-in-affective-modeling-large","title":"What's Next in Affective Modeling? Large Language Models","date":"2023-10-03","arxiv_id":"2310.18322","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-model-powered-smart-contract","slug":"large-language-model-powered-smart-contract","title":"Large Language Model-Powered Smart Contract Vulnerability Detection: New Perspectives","date":"2023-10-02","arxiv_id":"2310.01152","n_code_links":1,"syntology":null},{"paper":null,"slug":"loft-local-proxy-fine-tuning-for-improving","title":"LoFT: Local Proxy Fine-tuning For Improving Transferability Of Adversarial Attacks Against Large Language Model","date":"2023-10-02","arxiv_id":"2310.04445","n_code_links":0,"syntology":null},{"paper":"/paper/the-entity-deduction-arena-a-playground-for","slug":"the-entity-deduction-arena-a-playground-for","title":"Probing the Multi-turn Planning Capabilities of LLMs via 20 Question Games","date":"2023-10-02","arxiv_id":"2310.01468","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["apple/ml-entity-deduction-arena"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/tram-benchmarking-temporal-reasoning-for","slug":"tram-benchmarking-temporal-reasoning-for","title":"TRAM: Benchmarking Temporal Reasoning for Large Language Models","date":"2023-10-02","arxiv_id":"2310.00835","n_code_links":1,"syntology":null},{"paper":"/paper/ultrafeedback-boosting-language-models-with","slug":"ultrafeedback-boosting-language-models-with","title":"UltraFeedback: Boosting Language Models with Scaled AI Feedback","date":"2023-10-02","arxiv_id":"2310.01377","n_code_links":4,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["thunlp/ultrafeedback"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/who-is-chatgpt-benchmarking-llms","slug":"who-is-chatgpt-benchmarking-llms","title":"Who is ChatGPT? Benchmarking LLMs' Psychological Portrayal Using PsychoBench","date":"2023-10-02","arxiv_id":"2310.01386","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["cuhk-arise/psychobench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/booookscore-a-systematic-exploration-of-book","slug":"booookscore-a-systematic-exploration-of-book","title":"BooookScore: A systematic exploration of book-length summarization in the era of LLMs","date":"2023-10-01","arxiv_id":"2310.00785","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lilakk/booookscore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/tigerscore-towards-building-explainable","slug":"tigerscore-towards-building-explainable","title":"TIGERScore: Towards Building Explainable Metric for All Text Generation Tasks","date":"2023-10-01","arxiv_id":"2310.00752","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":8,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["TIGER-AI-Lab/TIGERScore"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/automatikz-text-guided-synthesis-of","slug":"automatikz-text-guided-synthesis-of","title":"AutomaTikZ: Text-Guided Synthesis of Scientific Vector Graphics with TikZ","date":"2023-09-30","arxiv_id":"2310.00367","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["potamides/automatikz"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"graph-neural-architecture-search-with-gpt-4","title":"Graph Neural Architecture Search with GPT-4","date":"2023-09-30","arxiv_id":"2310.01436","n_code_links":0,"syntology":null},{"paper":"/paper/measuring-value-understanding-in-language","slug":"measuring-value-understanding-in-language","title":"ValueDCG: Measuring Comprehensive Human Value Understanding Ability of Language Models","date":"2023-09-30","arxiv_id":"2310.00378","n_code_links":1,"syntology":null},{"paper":null,"slug":"upar-a-kantian-inspired-prompting-framework","title":"UPAR: A Kantian-Inspired Prompting Framework for Enhancing Large Language Model Capabilities","date":"2023-09-30","arxiv_id":"2310.01441","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-large-language-model-approach-to","title":"A Large Language Model Approach to Educational Survey Feedback Analysis","date":"2023-09-29","arxiv_id":"2309.17447","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-the-abilities-of-large-language","slug":"benchmarking-the-abilities-of-large-language","title":"Benchmarking the Abilities of Large Language Models for RDF Knowledge Graph Creation and Comprehension: How Well Do LLMs Speak Turtle?","date":"2023-09-29","arxiv_id":"2309.17122","n_code_links":3,"syntology":null},{"paper":"/paper/craft-customizing-llms-by-creating-and","slug":"craft-customizing-llms-by-creating-and","title":"CRAFT: Customizing LLMs by Creating and Retrieving from Specialized Toolsets","date":"2023-09-29","arxiv_id":"2309.17428","n_code_links":2,"syntology":{"ran":0,"of":5,"n_ran_checked":0,"n_instrument":0,"unverified":5,"pointer_only":5,"phrase":"0 ran · 5 unverified","official":{"repos":["lifan-yuan/craft"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"paper":"/paper/dyval-graph-informed-dynamic-evaluation-of","slug":"dyval-graph-informed-dynamic-evaluation-of","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","date":"2023-09-29","arxiv_id":"2309.17167","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-large-language-models-in-coding","slug":"enhancing-large-language-models-in-coding","title":"Enhancing Large Language Models in Coding Through Multi-Perspective Self-Consistency","date":"2023-09-29","arxiv_id":"2309.17272","n_code_links":1,"syntology":{"ran":8,"of":14,"n_ran_checked":7,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["skpig/MPSC"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/llm-deliberation-evaluating-llms-with","slug":"llm-deliberation-evaluating-llms-with","title":"Cooperation, Competition, and Maliciousness: LLM-Stakeholders Interactive Negotiation","date":"2023-09-29","arxiv_id":"2309.17234","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["s-abdelnabi/llm-deliberation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/scale-synergized-collaboration-of-asymmetric","slug":"scale-synergized-collaboration-of-asymmetric","title":"SCALE: Synergized Collaboration of Asymmetric Language Translation Engines","date":"2023-09-29","arxiv_id":"2309.17061","n_code_links":1,"syntology":null},{"paper":"/paper/socreval-large-language-models-with-the","slug":"socreval-large-language-models-with-the","title":"SocREval: Large Language Models with the Socratic Method for Reference-Free Reasoning Evaluation","date":"2023-09-29","arxiv_id":"2310.00074","n_code_links":1,"syntology":null},{"paper":null,"slug":"split-and-merge-aligning-position-biases-in","title":"Split and Merge: Aligning Position Biases in LLM-based Evaluators","date":"2023-09-29","arxiv_id":"2310.01432","n_code_links":0,"syntology":null},{"paper":"/paper/suspicion-agent-playing-imperfect-information","slug":"suspicion-agent-playing-imperfect-information","title":"Suspicion-Agent: Playing Imperfect Information Games with Theory of Mind Aware GPT-4","date":"2023-09-29","arxiv_id":"2309.17277","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cr-gjx/suspicion-agent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/tora-a-tool-integrated-reasoning-agent-for","slug":"tora-a-tool-integrated-reasoning-agent-for","title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","date":"2023-09-29","arxiv_id":"2309.17452","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":9,"n_instrument":2,"unverified":1,"pointer_only":9,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["microsoft/tora"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ae-gpt-using-large-language-models-to-extract","title":"AE-GPT: Using Large Language Models to Extract Adverse Events from Surveillance Reports-A Use Case with Influenza Vaccine Adverse Events","date":"2023-09-28","arxiv_id":"2309.16150","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-fathom-benchmarking-large-language-models","slug":"gpt-fathom-benchmarking-large-language-models","title":"GPT-Fathom: Benchmarking Large Language Models to Decipher the Evolutionary Path towards GPT-4 and Beyond","date":"2023-09-28","arxiv_id":"2309.16583","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["gpt-fathom/gpt-fathom"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/lawbench-benchmarking-legal-knowledge-of","slug":"lawbench-benchmarking-legal-knowledge-of","title":"LawBench: Benchmarking Legal Knowledge of Large Language Models","date":"2023-09-28","arxiv_id":"2309.16289","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["open-compass/lawbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"neuro-symbolic-reasoning-for-planning","title":"Neuro Symbolic Reasoning for Planning: Counterexample Guided Inductive Synthesis using Large Language Models and Satisfiability Solving","date":"2023-09-28","arxiv_id":"2309.16436","n_code_links":0,"syntology":null},{"paper":"/paper/chatcounselor-a-large-language-models-for","slug":"chatcounselor-a-large-language-models-for","title":"ChatCounselor: A Large Language Models for Mental Health Support","date":"2023-09-27","arxiv_id":"2309.15461","n_code_links":1,"syntology":null},{"paper":"/paper/benchmarking-large-language-models-on-cmexam-1","slug":"benchmarking-large-language-models-on-cmexam-1","title":"Benchmarking Large Language Models on CMExam - A comprehensive Chinese Medical Exam Dataset","date":"2023-09-26","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/capp-130-a-corpus-of-chinese-application","slug":"capp-130-a-corpus-of-chinese-application","title":"CAPP-130: A Corpus of Chinese Application Privacy Policy Summarization and Interpretation","date":"2023-09-26","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/question-answering-approach-to-evaluate-legal","slug":"question-answering-approach-to-evaluate-legal","title":"Question-Answering Approach to Evaluating Legal Summaries","date":"2023-09-26","arxiv_id":"2309.15016","n_code_links":1,"syntology":null},{"paper":"/paper/rankvicuna-zero-shot-listwise-document","slug":"rankvicuna-zero-shot-listwise-document","title":"RankVicuna: Zero-Shot Listwise Document Reranking with Open-Source Large Language Models","date":"2023-09-26","arxiv_id":"2309.15088","n_code_links":3,"syntology":null},{"paper":"/paper/supersonic-learning-to-generate-source-code","slug":"supersonic-learning-to-generate-source-code","title":"Supersonic: Learning to Generate Source Code Optimizations in C/C++","date":"2023-09-26","arxiv_id":"2309.14846","n_code_links":1,"syntology":null},{"paper":null,"slug":"visit-bench-a-dynamic-benchmark-for","title":"VisIT-Bench: A Dynamic Benchmark for Evaluating Instruction-Following Vision-and-Language Models","date":"2023-09-26","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"f3c8b5ae1ff9cbb43d1b18a0adb42de2daf4b03982b2ed3103ca9fa9aba94730","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}