{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/20","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":20,"pages_in_order":29,"rows_per_page":100,"rows":[1901,2000],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/19","next":"/method/gpt-4/papers/21","papers":[{"paper":null,"slug":"training-microrobots-to-swim-by-a-large","title":"Training microrobots to swim by a large language model","date":"2024-01-21","arxiv_id":"2402.00044","n_code_links":0,"syntology":null},{"paper":"/paper/badchain-backdoor-chain-of-thought-prompting","slug":"badchain-backdoor-chain-of-thought-prompting","title":"BadChain: Backdoor Chain-of-Thought Prompting for Large Language Models","date":"2024-01-20","arxiv_id":"2401.12242","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["django-jiang/badchain"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enhancing-large-language-models-for-clinical","title":"Enhancing Large Language Models for Clinical Decision Support by Incorporating Clinical Practice Guidelines","date":"2024-01-20","arxiv_id":"2401.11120","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-and-enhancing-large-language","title":"Evaluating and Enhancing Large Language Models Performance in Domain-specific Medicine: Osteoarthritis Management with DocOA","date":"2024-01-20","arxiv_id":"2401.12998","n_code_links":0,"syntology":null},{"paper":"/paper/inducing-high-energy-latency-of-large-vision","slug":"inducing-high-energy-latency-of-large-vision","title":"Inducing High Energy-Latency of Large Vision-Language Models with Verbose Images","date":"2024-01-20","arxiv_id":"2401.11170","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":0,"n_instrument":5,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["kuofenggao/verbose_images"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/mementos-a-comprehensive-benchmark-for","slug":"mementos-a-comprehensive-benchmark-for","title":"Mementos: A Comprehensive Benchmark for Multimodal Large Language Model Reasoning over Image Sequences","date":"2024-01-19","arxiv_id":"2401.10529","n_code_links":1,"syntology":null},{"paper":"/paper/mining-experimental-data-from-materials","slug":"mining-experimental-data-from-materials","title":"Mining experimental data from Materials Science literature with Large Language Models: an evaluation study","date":"2024-01-19","arxiv_id":"2401.11052","n_code_links":1,"syntology":null},{"paper":null,"slug":"beyond-reference-based-metrics-analyzing","title":"Beyond Traditional Benchmarks: Analyzing Behaviors of Open LLMs on Data-to-Text Generation","date":"2024-01-18","arxiv_id":"2401.10186","n_code_links":0,"syntology":null},{"paper":"/paper/chatqa-building-gpt-4-level-conversational-qa","slug":"chatqa-building-gpt-4-level-conversational-qa","title":"ChatQA: Surpassing GPT-4 on Conversational QA and RAG","date":"2024-01-18","arxiv_id":"2401.10225","n_code_links":0,"syntology":null},{"paper":"/paper/r-judge-benchmarking-safety-risk-awareness","slug":"r-judge-benchmarking-safety-risk-awareness","title":"R-Judge: Benchmarking Safety Risk Awareness for LLM Agents","date":"2024-01-18","arxiv_id":"2401.10019","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lordog/r-judge"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-rewarding-language-models","slug":"self-rewarding-language-models","title":"Self-Rewarding Language Models","date":"2024-01-18","arxiv_id":"2401.10020","n_code_links":3,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":null,"slug":"attackeval-how-to-evaluate-the-effectiveness","title":"AttackEval: How to Evaluate the Effectiveness of Jailbreak Attacking on Large Language Models","date":"2024-01-17","arxiv_id":"2401.09002","n_code_links":0,"syntology":null},{"paper":"/paper/augmenting-math-word-problems-via-iterative","slug":"augmenting-math-word-problems-via-iterative","title":"Augmenting Math Word Problems via Iterative Question Composing","date":"2024-01-17","arxiv_id":"2401.09003","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["iiis-ai/iterativequestioncomposing"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/bridging-research-and-readers-a-multi-modal","slug":"bridging-research-and-readers-a-multi-modal","title":"Bridging Research and Readers: A Multi-Modal Automated Academic Papers Interpretation System","date":"2024-01-17","arxiv_id":"2401.09150","n_code_links":1,"syntology":null},{"paper":"/paper/coco-is-all-you-need-for-visual-instruction","slug":"coco-is-all-you-need-for-visual-instruction","title":"COCO is \"ALL'' You Need for Visual Instruction Fine-tuning","date":"2024-01-17","arxiv_id":"2401.08968","n_code_links":0,"syntology":null},{"paper":"/paper/deciphering-textual-authenticity-a","slug":"deciphering-textual-authenticity-a","title":"Deciphering Textual Authenticity: A Generalized Strategy through the Lens of Large Language Semantics for Detecting Human vs. Machine-Generated Text","date":"2024-01-17","arxiv_id":"2401.09407","n_code_links":1,"syntology":null},{"paper":null,"slug":"from-user-surveys-to-telemetry-driven-agents","title":"From User Surveys to Telemetry-Driven AI Agents: Exploring the Potential of Personalized Productivity Solutions","date":"2024-01-17","arxiv_id":"2401.08960","n_code_links":0,"syntology":null},{"paper":null,"slug":"impact-of-large-language-model-assistance-on","title":"Impact of Large Language Model Assistance on Patients Reading Clinical Notes: A Mixed-Methods Study","date":"2024-01-17","arxiv_id":"2401.09637","n_code_links":0,"syntology":null},{"paper":"/paper/stuck-in-the-quicksand-of-numeracy-far-from","slug":"stuck-in-the-quicksand-of-numeracy-far-from","title":"Evaluating LLMs' Mathematical and Coding Competency through Ontology-guided Interventions","date":"2024-01-17","arxiv_id":"2401.09395","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["declare-lab/llm-reasoningtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/code-generation-with-alphacodium-from-prompt","slug":"code-generation-with-alphacodium-from-prompt","title":"Code Generation with AlphaCodium: From Prompt Engineering to Flow Engineering","date":"2024-01-16","arxiv_id":"2401.08500","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["codium-ai/alphacodium"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/contrastive-preference-optimization-pushing","slug":"contrastive-preference-optimization-pushing","title":"Contrastive Preference Optimization: Pushing the Boundaries of LLM Performance in Machine Translation","date":"2024-01-16","arxiv_id":"2401.08417","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":4,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fe1ixxu/alma"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/emollms-a-series-of-emotional-large-language","slug":"emollms-a-series-of-emotional-large-language","title":"EmoLLMs: A Series of Emotional Large Language Models and Annotation Tools for Comprehensive Affective Analysis","date":"2024-01-16","arxiv_id":"2401.08508","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":0,"n_instrument":3,"unverified":3,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["lzw108/emollms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/forging-vision-foundation-models-for","slug":"forging-vision-foundation-models-for","title":"Forging Vision Foundation Models for Autonomous Driving: Challenges, Methodologies, and Opportunities","date":"2024-01-16","arxiv_id":"2401.08045","n_code_links":1,"syntology":null},{"paper":"/paper/mario-math-reasoning-with-code-interpreter","slug":"mario-math-reasoning-with-code-interpreter","title":"MARIO: MAth Reasoning with code Interpreter Output -- A Reproducible Pipeline","date":"2024-01-16","arxiv_id":"2401.08190","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mario-math-reasoning/mario"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mmtom-qa-multimodal-theory-of-mind-question","slug":"mmtom-qa-multimodal-theory-of-mind-question","title":"MMToM-QA: Multimodal Theory of Mind Question Answering","date":"2024-01-16","arxiv_id":"2401.08743","n_code_links":1,"syntology":null},{"paper":null,"slug":"rag-vs-fine-tuning-pipelines-tradeoffs-and-a","title":"RAG vs Fine-tuning: Pipelines, Tradeoffs, and a Case Study on Agriculture","date":"2024-01-16","arxiv_id":"2401.08406","n_code_links":0,"syntology":null},{"paper":"/paper/rotbench-a-multi-level-benchmark-for","slug":"rotbench-a-multi-level-benchmark-for","title":"RoTBench: A Multi-Level Benchmark for Evaluating the Robustness of Large Language Models in Tool Learning","date":"2024-01-16","arxiv_id":"2401.08326","n_code_links":1,"syntology":{"ran":10,"of":15,"n_ran_checked":10,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["junjie-ye/rotbench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-novel-approach-for-automatic-program-repair","slug":"a-novel-approach-for-automatic-program-repair","title":"A Novel Approach for Automatic Program Repair using Round-Trip Translation with Large Language Models","date":"2024-01-15","arxiv_id":"2401.07994","n_code_links":1,"syntology":null},{"paper":null,"slug":"consolidating-trees-of-robotic-plans","title":"Consolidating Trees of Robotic Plans Generated Using Large Language Models to Improve Reliability","date":"2024-01-15","arxiv_id":"2401.07868","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploiting-gpt-4-vision-for-zero-shot-point","title":"Exploiting GPT-4 Vision for Zero-shot Point Cloud Understanding","date":"2024-01-15","arxiv_id":"2401.07572","n_code_links":0,"syntology":null},{"paper":null,"slug":"mapgpt-map-guided-prompting-for-unified","title":"MapGPT: Map-Guided Prompting with Adaptive Path Planning for Vision-and-Language Navigation","date":"2024-01-14","arxiv_id":"2401.07314","n_code_links":0,"syntology":null},{"paper":null,"slug":"streamlining-the-selection-phase-of","title":"Streamlining the Selection Phase of Systematic Literature Reviews (SLRs) Using AI-Enabled GPT-4 Assistant API","date":"2024-01-14","arxiv_id":"2402.18582","n_code_links":0,"syntology":null},{"paper":"/paper/a-novel-multi-stage-prompting-approach-for","slug":"a-novel-multi-stage-prompting-approach-for","title":"A Novel Multi-Stage Prompting Approach for Language Agnostic MCQ Generation using GPT","date":"2024-01-13","arxiv_id":"2401.07098","n_code_links":1,"syntology":null},{"paper":null,"slug":"assessing-large-language-models-in-mechanical","title":"Assessing Large Language Models in Mechanical Engineering Education: A Study on Mechanics-Focused Conceptual Understanding","date":"2024-01-13","arxiv_id":"2401.12983","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-distillation-for-closed-source","title":"Knowledge Distillation of Black-Box Large Language Models","date":"2024-01-13","arxiv_id":"2401.07013","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-survey-on-the-applications-of-frontier-ai","title":"A Survey on the Applications of Frontier AI, Foundation Models, and Large Language Models to Intelligent Transportation Systems","date":"2024-01-12","arxiv_id":"2401.06831","n_code_links":0,"syntology":null},{"paper":null,"slug":"adapting-large-language-models-for-document","title":"Adapting Large Language Models for Document-Level Machine Translation","date":"2024-01-12","arxiv_id":"2401.06468","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-gpt-4-and-open-source-language","title":"Comparing GPT-4 and Open-Source Language Models in Misinformation Mitigation","date":"2024-01-12","arxiv_id":"2401.06920","n_code_links":0,"syntology":null},{"paper":"/paper/few-shot-detection-of-machine-generated-text","slug":"few-shot-detection-of-machine-generated-text","title":"Few-Shot Detection of Machine-Generated Text using Style Representations","date":"2024-01-12","arxiv_id":"2401.06712","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["llnl/luar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fine-grained-hallucination-detection-and","title":"Fine-grained Hallucination Detection and Editing for Language Models","date":"2024-01-12","arxiv_id":"2401.06855","n_code_links":0,"syntology":null},{"paper":"/paper/from-automation-to-augmentation-large","slug":"from-automation-to-augmentation-large","title":"Human-AI Collaborative Essay Scoring: A Dual-Process Framework with LLMs","date":"2024-01-12","arxiv_id":"2401.06431","n_code_links":1,"syntology":null},{"paper":"/paper/health-llm-large-language-models-for-health","slug":"health-llm-large-language-models-for-health","title":"Health-LLM: Large Language Models for Health Prediction via Wearable Sensor Data","date":"2024-01-12","arxiv_id":"2401.06866","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mitmedialab/health-llm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-johnny-can-persuade-llms-to-jailbreak","slug":"how-johnny-can-persuade-llms-to-jailbreak","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","date":"2024-01-12","arxiv_id":"2401.06373","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chats-lab/persuasive_jailbreaker"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/pizzacommonsense-learning-to-model","slug":"pizzacommonsense-learning-to-model","title":"PizzaCommonSense: Learning to Model Commonsense Reasoning about Intermediate Steps in Cooking Recipes","date":"2024-01-12","arxiv_id":"2401.06930","n_code_links":1,"syntology":null},{"paper":null,"slug":"autocompletion-of-chief-complaints-in-the","title":"Autocompletion of Chief Complaints in the Electronic Health Records using Large Language Models","date":"2024-01-11","arxiv_id":"2401.06088","n_code_links":0,"syntology":null},{"paper":null,"slug":"evidence-to-generate-e2g-a-single-agent-two","title":"Evidence to Generate (E2G): A Single-agent Two-step Prompting for Context Grounded and Retrieval Augmented Reasoning","date":"2024-01-11","arxiv_id":"2401.05787","n_code_links":0,"syntology":null},{"paper":null,"slug":"mutation-based-consistency-testing-for","title":"Mutation-based Consistency Testing for Evaluating the Code Understanding Capability of LLMs","date":"2024-01-11","arxiv_id":"2401.05940","n_code_links":0,"syntology":null},{"paper":"/paper/the-benefits-of-a-concise-chain-of-thought-on","slug":"the-benefits-of-a-concise-chain-of-thought-on","title":"The Benefits of a Concise Chain of Thought on Problem-Solving in Large Language Models","date":"2024-01-11","arxiv_id":"2401.05618","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["matthewrenze/jhu-concise-cot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cadgpt-harnessing-natural-language-processing","title":"CADgpt: Harnessing Natural Language Processing for 3D Modelling to Enhance Computer-Aided Design Workflows","date":"2024-01-10","arxiv_id":"2401.05476","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-ai-write-classical-chinese-poetry-like","title":"Can AI Write Classical Chinese Poetry like Humans? An Empirical Study Inspired by Turing Test","date":"2024-01-10","arxiv_id":"2401.04952","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-sharing-in-manufacturing-using","title":"Knowledge Sharing in Manufacturing using Large Language Models: User Evaluation and Model Benchmarking","date":"2024-01-10","arxiv_id":"2401.05200","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-print-debugging-to-improve-code","title":"Leveraging Print Debugging to Improve Code Generation in Large Language Models","date":"2024-01-10","arxiv_id":"2401.05319","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-optimizing-rag-for","title":"Reinforcement Learning for Optimizing RAG for Domain Chatbots","date":"2024-01-10","arxiv_id":"2401.06800","n_code_links":0,"syntology":null},{"paper":null,"slug":"arabic-text-diacritization-in-the-age-of","title":"Arabic Text Diacritization In The Age Of Transfer Learning: Token Classification Is All You Need","date":"2024-01-09","arxiv_id":"2401.04848","n_code_links":0,"syntology":null},{"paper":"/paper/debugbench-evaluating-debugging-capability-of","slug":"debugbench-evaluating-debugging-capability-of","title":"DebugBench: Evaluating Debugging Capability of Large Language Models","date":"2024-01-09","arxiv_id":"2401.04621","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":4,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["thunlp/debugbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/informed-ai-regulation-comparing-the-ethical","slug":"informed-ai-regulation-comparing-the-ethical","title":"Informed AI Regulation: Comparing the Ethical Frameworks of Leading LLM Chatbots Using an Ethics-Based Audit to Assess Moral Reasoning and Normative Values","date":"2024-01-09","arxiv_id":"2402.01651","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-philosophical-introduction-to-language","title":"A Philosophical Introduction to Language Models -- Part I: Continuity With Classic Debates","date":"2024-01-08","arxiv_id":"2401.03910","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-large-language-models-beat-wall-street","title":"Can Large Language Models Beat Wall Street? Unveiling the Potential of AI in Stock Selection","date":"2024-01-08","arxiv_id":"2401.03737","n_code_links":0,"syntology":null},{"paper":null,"slug":"distortions-in-judged-spatial-relations-in","title":"Distortions in Judged Spatial Relations in Large Language Models","date":"2024-01-08","arxiv_id":"2401.04218","n_code_links":0,"syntology":null},{"paper":"/paper/llm4plc-harnessing-large-language-models-for","slug":"llm4plc-harnessing-large-language-models-for","title":"LLM4PLC: Harnessing Large Language Models for Verifiable Programming of PLCs in Industrial Control Systems","date":"2024-01-08","arxiv_id":"2401.05443","n_code_links":1,"syntology":null},{"paper":"/paper/marg-multi-agent-review-generation-for","slug":"marg-multi-agent-review-generation-for","title":"MARG: Multi-Agent Review Generation for Scientific Papers","date":"2024-01-08","arxiv_id":"2401.04259","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["allenai/marg-reviewer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"why-solving-multi-agent-path-finding-with","title":"Why Solving Multi-agent Path Finding with Large Language Model has not Succeeded Yet","date":"2024-01-08","arxiv_id":"2401.03630","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-generative-ai-and-chatgpt-outperform","title":"Can generative AI and ChatGPT outperform humans on cognitive-demanding problem-solving tasks in science?","date":"2024-01-07","arxiv_id":"2401.15081","n_code_links":0,"syntology":null},{"paper":"/paper/escalation-risks-from-language-models-in","slug":"escalation-risks-from-language-models-in","title":"Escalation Risks from Language Models in Military and Diplomatic Decision-Making","date":"2024-01-07","arxiv_id":"2401.03408","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jprivera44/EscalAItion"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/infobench-evaluating-instruction-following","slug":"infobench-evaluating-instruction-following","title":"InFoBench: Evaluating Instruction Following Ability in Large Language Models","date":"2024-01-07","arxiv_id":"2401.03601","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qinyiwei/infobench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-leveraging-large-language-models-for","title":"On Leveraging Large Language Models for Enhancing Entity Resolution: A Cost-efficient Approach","date":"2024-01-07","arxiv_id":"2401.03426","n_code_links":0,"syntology":null},{"paper":null,"slug":"token-free-llms-can-generate-chinese","title":"CharPoet: A Chinese Classical Poetry Generation System Based on Token-free LLM","date":"2024-01-07","arxiv_id":"2401.03512","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-large-language-models-to-assess-tutors","title":"Using Large Language Models to Assess Tutors' Performance in Reacting to Students Making Math Errors","date":"2024-01-06","arxiv_id":"2401.03238","n_code_links":0,"syntology":null},{"paper":"/paper/cruxeval-a-benchmark-for-code-reasoning","slug":"cruxeval-a-benchmark-for-code-reasoning","title":"CRUXEval: A Benchmark for Code Reasoning, Understanding and Execution","date":"2024-01-05","arxiv_id":"2401.03065","n_code_links":1,"syntology":null},{"paper":null,"slug":"from-llm-to-conversational-agent-a-memory","title":"From LLM to Conversational Agent: A Memory Enhanced Architecture with Fine-Tuning of Large Language Models","date":"2024-01-05","arxiv_id":"2401.02777","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-large-language-models-are","title":"Natural Language Programming in Medicine: Administering Evidence Based Clinical Workflows with Autonomous Agents Powered by Generative Large Language Models","date":"2024-01-05","arxiv_id":"2401.02851","n_code_links":0,"syntology":null},{"paper":"/paper/pefomed-parameter-efficient-fine-tuning-on","slug":"pefomed-parameter-efficient-fine-tuning-on","title":"PeFoMed: Parameter Efficient Fine-tuning of Multimodal Large Language Models for Medical Imaging","date":"2024-01-05","arxiv_id":"2401.02797","n_code_links":1,"syntology":null},{"paper":null,"slug":"blar-sql-faster-stronger-smaller-nl2sql","title":"Blar-SQL: Faster, Stronger, Smaller NL2SQL","date":"2024-01-04","arxiv_id":"2401.02997","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-boundary-of-gpt-4v-on-marine","title":"Exploring Boundary of GPT-4V on Marine Analysis: A Preliminary Case Study","date":"2024-01-04","arxiv_id":"2401.02147","n_code_links":0,"syntology":null},{"paper":"/paper/aigcbench-comprehensive-evaluation-of-image","slug":"aigcbench-comprehensive-evaluation-of-image","title":"AIGCBench: Comprehensive Evaluation of Image-to-Video Content Generated by AI","date":"2024-01-03","arxiv_id":"2401.01651","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["benchcouncil/aigcbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"astrollama-chat-scaling-astrollama-with","title":"AstroLLaMA-Chat: Scaling AstroLLaMA with Conversational and Diverse Datasets","date":"2024-01-03","arxiv_id":"2401.01916","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-4v-ision-is-a-generalist-web-agent-if","slug":"gpt-4v-ision-is-a-generalist-web-agent-if","title":"GPT-4V(ision) is a Generalist Web Agent, if Grounded","date":"2024-01-03","arxiv_id":"2401.01614","n_code_links":1,"syntology":{"ran":13,"of":14,"n_ran_checked":12,"n_instrument":1,"unverified":1,"pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["osu-nlp-group/seeact"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-model-capabilities-in","slug":"large-language-model-capabilities-in","title":"Large Language Model Capabilities in Perioperative Risk Prediction and Prognostication","date":"2024-01-03","arxiv_id":"2401.01620","n_code_links":1,"syntology":null},{"paper":null,"slug":"team-ielab-at-trec-clinical-trial-track-2023","title":"Team IELAB at TREC Clinical Trial Track 2023: Enhancing Clinical Trial Retrieval with Neural Rankers and Large Language Models","date":"2024-01-03","arxiv_id":"2401.01566","n_code_links":0,"syntology":null},{"paper":"/paper/charactereval-a-chinese-benchmark-for-role","slug":"charactereval-a-chinese-benchmark-for-role","title":"CharacterEval: A Chinese Benchmark for Role-Playing Conversational Agent Evaluation","date":"2024-01-02","arxiv_id":"2401.01275","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["morecry/charactereval"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"evaluating-large-language-models-on-the-gmat","title":"Evaluating Large Language Models on the GMAT: Implications for the Future of Business Education","date":"2024-01-02","arxiv_id":"2401.02985","n_code_links":0,"syntology":null},{"paper":null,"slug":"identification-of-regulatory-requirements","title":"Identification of Regulatory Requirements Relevant to Business Processes: A Comparative Study on Generative AI, Embedding-based Ranking, Crowd and Expert-driven Methods","date":"2024-01-02","arxiv_id":"2401.02986","n_code_links":0,"syntology":null},{"paper":"/paper/self-play-fine-tuning-converts-weak-language","slug":"self-play-fine-tuning-converts-weak-language","title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","date":"2024-01-02","arxiv_id":"2401.01335","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["uclaml/SPIN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"uncertainty-resolution-in-misinformation","title":"Uncertainty Resolution in Misinformation Detection","date":"2024-01-02","arxiv_id":"2401.01197","n_code_links":0,"syntology":null},{"paper":"/paper/a-b-b-a-triggering-logical-reasoning-failures","slug":"a-b-b-a-triggering-logical-reasoning-failures","title":"LogicAsker: Evaluating and Improving the Logical Reasoning Ability of Large Language Models","date":"2024-01-01","arxiv_id":"2401.00757","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yxwan123/logicasker"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enhanced-motion-text-alignment-for-image-to","title":"Enhanced Motion-Text Alignment for Image-to-Video Transfer Learning","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"taking-the-next-step-with-generative","title":"Taking the Next Step with Generative Artificial Intelligence: The Transformative Role of Multimodal Large Language Models in Science Education","date":"2024-01-01","arxiv_id":"2401.00832","n_code_links":0,"syntology":null},{"paper":"/paper/ragtruth-a-hallucination-corpus-for","slug":"ragtruth-a-hallucination-corpus-for","title":"RAGTruth: A Hallucination Corpus for Developing Trustworthy Retrieval-Augmented Language Models","date":"2023-12-31","arxiv_id":"2401.00396","n_code_links":3,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["particlemedia/ragtruth"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/red-teaming-for-large-language-models-at","slug":"red-teaming-for-large-language-models-at","title":"Red Teaming for Large Language Models At Scale: Tackling Hallucinations on Mathematics Tasks","date":"2023-12-30","arxiv_id":"2401.00290","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficacy-of-utilizing-large-language-models","title":"Efficacy of Utilizing Large Language Models to Detect Public Threat Posted Online","date":"2023-12-29","arxiv_id":"2401.02974","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-quantitative-reasoning-skills-of","title":"Enhancing Quantitative Reasoning Skills of Large Language Models through Dimension Perception","date":"2023-12-29","arxiv_id":"2312.17532","n_code_links":0,"syntology":null},{"paper":"/paper/challenge-llms-to-reason-about-reasoning-a","slug":"challenge-llms-to-reason-about-reasoning-a","title":"MR-GSM8K: A Meta-Reasoning Benchmark for Large Language Model Evaluation","date":"2023-12-28","arxiv_id":"2312.17080","n_code_links":2,"syntology":null},{"paper":"/paper/gitagent-facilitating-autonomous-agent-with","slug":"gitagent-facilitating-autonomous-agent-with","title":"Enhancing Open-Domain Task-Solving Capability of LLMs via Autonomous Tool Integration from GitHub","date":"2023-12-28","arxiv_id":"2312.17294","n_code_links":1,"syntology":null},{"paper":"/paper/roleeval-a-bilingual-role-evaluation","slug":"roleeval-a-bilingual-role-evaluation","title":"RoleEval: A Bilingual Role Evaluation Benchmark for Large Language Models","date":"2023-12-26","arxiv_id":"2312.16132","n_code_links":1,"syntology":null},{"paper":"/paper/secqa-a-concise-question-answering-dataset","slug":"secqa-a-concise-question-answering-dataset","title":"SecQA: A Concise Question-Answering Dataset for Evaluating Large Language Models in Computer Security","date":"2023-12-26","arxiv_id":"2312.15838","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zefang-liu/lm-evaluation-harness"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"esgreveal-an-llm-based-approach-for","title":"ESGReveal: An LLM-based approach for extracting structured data from ESG reports","date":"2023-12-25","arxiv_id":"2312.17264","n_code_links":0,"syntology":null},{"paper":null,"slug":"iqagpt-image-quality-assessment-with-vision","title":"IQAGPT: Image Quality Assessment with Vision-language and ChatGPT Models","date":"2023-12-25","arxiv_id":"2312.15663","n_code_links":0,"syntology":null},{"paper":null,"slug":"deap-design-space-exploration-for-dnn","title":"DEAP: Design Space Exploration for DNN Accelerator Parallelism","date":"2023-12-24","arxiv_id":"2312.15388","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-llm-agents-exhibit-social-behavior","title":"Do LLM Agents Exhibit Social Behavior?","date":"2023-12-23","arxiv_id":"2312.15198","n_code_links":0,"syntology":null},{"paper":null,"slug":"evolving-large-language-model-assistant-with","title":"Personalized Large Language Model Assistant with Evolving Conditional Memory","date":"2023-12-22","arxiv_id":"2312.17257","n_code_links":0,"syntology":null}],"record_sha256":"2fdaf0dae80e56caa3086624d2cc4b07693849ab36bd1f6d91abd2180ed3cd42","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}