{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/10","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":10,"pages_in_order":29,"rows_per_page":100,"rows":[901,1000],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/9","next":"/method/gpt-4/papers/11","papers":[{"paper":"/paper/step-by-step-reasoning-to-solve-grid-puzzles","slug":"step-by-step-reasoning-to-solve-grid-puzzles","title":"Step-by-Step Reasoning to Solve Grid Puzzles: Where do LLMs Falter?","date":"2024-07-20","arxiv_id":"2407.14790","n_code_links":1,"syntology":null},{"paper":null,"slug":"travellm-could-you-plan-my-new-public-transit","title":"TraveLLM: Could you plan my new public transit route in face of a network disruption?","date":"2024-07-20","arxiv_id":"2407.14926","n_code_links":0,"syntology":null},{"paper":null,"slug":"domain-specific-pretraining-of-language","title":"Domain-Specific Pretraining of Language Models: A Comparative Study in the Medical Field","date":"2024-07-19","arxiv_id":"2407.14076","n_code_links":0,"syntology":null},{"paper":null,"slug":"hecix-integrating-knowledge-graphs-and-large","title":"HeCiX: Integrating Knowledge Graphs and Large Language Models for Biomedical Research","date":"2024-07-19","arxiv_id":"2407.14030","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-retrieval-in-sponsored-search-by","title":"Improving Retrieval in Sponsored Search by Leveraging Query Context Signals","date":"2024-07-19","arxiv_id":"2407.14346","n_code_links":0,"syntology":null},{"paper":null,"slug":"lapis-language-model-augmented-police","title":"LAPIS: Language Model-Augmented Police Investigation System","date":"2024-07-19","arxiv_id":"2407.20248","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-left-right-and-center-assessing-gpt-s","title":"LLMs left, right, and center: Assessing GPT's capabilities to label political bias from web domains","date":"2024-07-19","arxiv_id":"2407.14344","n_code_links":0,"syntology":null},{"paper":null,"slug":"sqlfuse-enhancing-text-to-sql-performance","title":"SQLfuse: Enhancing Text-to-SQL Performance through Comprehensive LLM Synergy","date":"2024-07-19","arxiv_id":"2407.14568","n_code_links":0,"syntology":null},{"paper":"/paper/can-open-source-llms-compete-with-commercial","slug":"can-open-source-llms-compete-with-commercial","title":"Can Open-Source LLMs Compete with Commercial Models? Exploring the Few-Shot Performance of Current GPT Models in Biomedical Tasks","date":"2024-07-18","arxiv_id":"2407.13511","n_code_links":1,"syntology":null},{"paper":null,"slug":"dream-a-biomedical-data-driven-self-evolving","title":"Autonomous self-evolving research on biomedical data: the DREAM paradigm","date":"2024-07-18","arxiv_id":"2407.13637","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-sarcasm-detection-a-step-by-step-reasoning","title":"Is Sarcasm Detection A Step-by-Step Reasoning Process in Large Language Models?","date":"2024-07-17","arxiv_id":"2407.12725","n_code_links":0,"syntology":null},{"paper":"/paper/does-refusal-training-in-llms-generalize-to","slug":"does-refusal-training-in-llms-generalize-to","title":"Does Refusal Training in LLMs Generalize to the Past Tense?","date":"2024-07-16","arxiv_id":"2407.11969","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tml-epfl/llm-past-tense"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ecoh-turn-level-coherence-evaluation-for","slug":"ecoh-turn-level-coherence-evaluation-for","title":"ECoh: Turn-level Coherence Evaluation for Multilingual Dialogues","date":"2024-07-16","arxiv_id":"2407.11660","n_code_links":1,"syntology":null},{"paper":null,"slug":"educational-personalized-learning-path","title":"Educational Personalized Learning Path Planning with Large Language Models","date":"2024-07-16","arxiv_id":"2407.11773","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-assisted-annotation-of-rhetorical-and","title":"GPT Assisted Annotation of Rhetorical and Linguistic Features for Interpretable Propaganda Technique Detection in News Text","date":"2024-07-16","arxiv_id":"2407.11827","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-as-misleading","title":"Large Language Models as Misleading Assistants in Conversation","date":"2024-07-16","arxiv_id":"2407.11789","n_code_links":0,"syntology":null},{"paper":"/paper/lofti-localization-and-factuality-transfer-to","slug":"lofti-localization-and-factuality-transfer-to","title":"LoFTI: Localization and Factuality Transfer to Indian Locales","date":"2024-07-16","arxiv_id":"2407.11833","n_code_links":1,"syntology":null},{"paper":null,"slug":"review-feedback-reason-refer-a-novel","title":"ReFeR: Improving Evaluation and Reasoning through Hierarchy of Models","date":"2024-07-16","arxiv_id":"2407.12877","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-generative-artificial-intelligence","title":"Beyond Generative Artificial Intelligence: Roadmap for Natural Language Generation","date":"2024-07-15","arxiv_id":"2407.10554","n_code_links":0,"syntology":null},{"paper":null,"slug":"clave-an-adaptive-framework-for-evaluating","title":"CLAVE: An Adaptive Framework for Evaluating Values of LLM Generated Responses","date":"2024-07-15","arxiv_id":"2407.10725","n_code_links":0,"syntology":null},{"paper":null,"slug":"empowering-llms-for-verilog-generation","title":"CodeV: Empowering LLMs with HDL Generation through Multi-Level Summarization","date":"2024-07-15","arxiv_id":"2407.10424","n_code_links":0,"syntology":null},{"paper":null,"slug":"foundational-autoraters-taming-large-language","title":"Foundational Autoraters: Taming Large Language Models for Better Automatic Evaluation","date":"2024-07-15","arxiv_id":"2407.10817","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-llm-respondents-for-item","title":"Leveraging LLM-Respondents for Item Evaluation: a Psychometric Analysis","date":"2024-07-15","arxiv_id":"2407.10899","n_code_links":0,"syntology":null},{"paper":"/paper/sibyl-simple-yet-effective-agent-framework","slug":"sibyl-simple-yet-effective-agent-framework","title":"Sibyl: Simple yet Effective Agent Framework for Complex Real-world Reasoning","date":"2024-07-15","arxiv_id":"2407.10718","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ag2s1/sibyl-system"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/chatlogic-integrating-logic-programming-with","slug":"chatlogic-integrating-logic-programming-with","title":"ChatLogic: Integrating Logic Programming with Large Language Models for Multi-Step Reasoning","date":"2024-07-14","arxiv_id":"2407.10162","n_code_links":1,"syntology":null},{"paper":"/paper/benchmarking-llms-for-optimization-modeling","slug":"benchmarking-llms-for-optimization-modeling","title":"OptiBench Meets ReSocratic: Measure and Improve LLMs for Optimization Modeling","date":"2024-07-13","arxiv_id":"2407.09887","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yangzhch6/ReSocratic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"causality-extraction-from-medical-text-using","title":"Causality extraction from medical text using Large Language Models (LLMs)","date":"2024-07-13","arxiv_id":"2407.10020","n_code_links":0,"syntology":null},{"paper":null,"slug":"cohesive-conversations-enhancing-authenticity","title":"Cohesive Conversations: Enhancing Authenticity in Multi-Agent Simulated Dialogues","date":"2024-07-13","arxiv_id":"2407.09897","n_code_links":0,"syntology":null},{"paper":"/paper/llm-collaboration-on-automatic-science","slug":"llm-collaboration-on-automatic-science","title":"LLM-Collaboration on Automatic Science Journalism for the General Audience","date":"2024-07-13","arxiv_id":"2407.09756","n_code_links":1,"syntology":null},{"paper":"/paper/benchmarking-language-model-creativity-a-case","slug":"benchmarking-language-model-creativity-a-case","title":"Benchmarking Language Model Creativity: A Case Study on Code Generation","date":"2024-07-12","arxiv_id":"2407.09007","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["JHU-CLSP/NeoCoder"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/leveraging-large-language-models-for-nano","slug":"leveraging-large-language-models-for-nano","title":"Leveraging large language models for nano synthesis mechanism explanation: solid foundations or mere conjectures?","date":"2024-07-12","arxiv_id":"2407.08922","n_code_links":1,"syntology":null},{"paper":"/paper/refuse-whenever-you-feel-unsafe-improving","slug":"refuse-whenever-you-feel-unsafe-improving","title":"Refuse Whenever You Feel Unsafe: Improving Safety in LLMs via Decoupled Refusal Training","date":"2024-07-12","arxiv_id":"2407.09121","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["robustnlp/derta"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"self-evolving-gpt-a-lifelong-autonomous","title":"Self-Evolving GPT: A Lifelong Autonomous Experiential Learner","date":"2024-07-12","arxiv_id":"2407.08937","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-prompt-tuning-enable-autonomous-role","title":"Self-Prompt Tuning: Enable Autonomous Role-Playing in LLMs","date":"2024-07-12","arxiv_id":"2407.08995","n_code_links":0,"syntology":null},{"paper":"/paper/show-don-t-tell-evaluating-large-language","slug":"show-don-t-tell-evaluating-large-language","title":"Show, Don't Tell: Evaluating Large Language Models Beyond Textual Understanding with ChildPlay","date":"2024-07-12","arxiv_id":"2407.11068","n_code_links":1,"syntology":null},{"paper":null,"slug":"telecomgpt-a-framework-to-build-telecom","title":"TelecomGPT: A Framework to Build Telecom-Specfic Large Language Models","date":"2024-07-12","arxiv_id":"2407.09424","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-two-sides-of-the-coin-hallucination","title":"The Two Sides of the Coin: Hallucination Generation and Detection with LLMs as Evaluators for LLMs","date":"2024-07-12","arxiv_id":"2407.09152","n_code_links":0,"syntology":null},{"paper":null,"slug":"converging-paradigms-the-synergy-of-symbolic","title":"Converging Paradigms: The Synergy of Symbolic and Connectionist AI in LLM-Empowered Autonomous Agents","date":"2024-07-11","arxiv_id":"2407.08516","n_code_links":0,"syntology":null},{"paper":null,"slug":"fault-diagnosis-in-power-grids-with-large","title":"Fault Diagnosis in Power Grids with Large Language Model","date":"2024-07-11","arxiv_id":"2407.08836","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-is-judged-more-human-than-humans-in","title":"GPT-4 is judged more human than humans in displaced and inverted Turing tests","date":"2024-07-11","arxiv_id":"2407.08853","n_code_links":0,"syntology":null},{"paper":"/paper/gta-a-benchmark-for-general-tool-agents","slug":"gta-a-benchmark-for-general-tool-agents","title":"GTA: A Benchmark for General Tool Agents","date":"2024-07-11","arxiv_id":"2407.08713","n_code_links":1,"syntology":null},{"paper":null,"slug":"skywork-math-data-scaling-laws-for","title":"Skywork-Math: Data Scaling Laws for Mathematical Reasoning in Large Language Models -- The Story Goes On","date":"2024-07-11","arxiv_id":"2407.08348","n_code_links":0,"syntology":null},{"paper":"/paper/arabic-automatic-story-generation-with-large","slug":"arabic-automatic-story-generation-with-large","title":"Arabic Automatic Story Generation with Large Language Models","date":"2024-07-10","arxiv_id":"2407.07551","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-large-language-models-with-grid","slug":"evaluating-large-language-models-with-grid","title":"Evaluating Large Language Models with Grid-Based Game Competitions: An Extensible LLM Benchmark and Leaderboard","date":"2024-07-10","arxiv_id":"2407.07796","n_code_links":1,"syntology":null},{"paper":"/paper/litsearch-a-retrieval-benchmark-for","slug":"litsearch-a-retrieval-benchmark-for","title":"LitSearch: A Retrieval Benchmark for Scientific Literature Search","date":"2024-07-10","arxiv_id":"2407.18940","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["princeton-nlp/litsearch"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mixsumm-topic-based-data-augmentation-using","title":"A Guide To Effectively Leveraging LLMs for Low-Resource Text Summarization: Data Augmentation and Semi-supervised Approaches","date":"2024-07-10","arxiv_id":"2407.07341","n_code_links":0,"syntology":null},{"paper":null,"slug":"probability-of-differentiation-reveals","title":"Probability of Differentiation Reveals Brittleness of Homogeneity Bias in GPT-4","date":"2024-07-10","arxiv_id":"2407.07329","n_code_links":0,"syntology":null},{"paper":null,"slug":"rag-vs-long-context-examining-frontier-large","title":"Examining Long-Context Large Language Models for Environmental Review Document Comprehension","date":"2024-07-10","arxiv_id":"2407.07321","n_code_links":0,"syntology":null},{"paper":null,"slug":"teaching-transformers-causal-reasoning","title":"Teaching Transformers Causal Reasoning through Axiomatic Training","date":"2024-07-10","arxiv_id":"2407.07612","n_code_links":0,"syntology":null},{"paper":null,"slug":"worldapis-the-world-is-worth-how-many-apis-a","title":"WorldAPIs: The World Is Worth How Many APIs? A Thought Experiment","date":"2024-07-10","arxiv_id":"2407.07778","n_code_links":0,"syntology":null},{"paper":"/paper/automated-peer-reviewing-in-paper-sea","slug":"automated-peer-reviewing-in-paper-sea","title":"Automated Peer Reviewing in Paper SEA: Standardization, Evaluation, and Analysis","date":"2024-07-09","arxiv_id":"2407.12857","n_code_links":1,"syntology":{"ran":7,"of":14,"n_ran_checked":7,"n_instrument":0,"unverified":7,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["ecnu-sea/SEA"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"convnlp-image-based-ai-text-detection","title":"ConvNLP: Image-based AI Text Detection","date":"2024-07-09","arxiv_id":"2407.07225","n_code_links":0,"syntology":null},{"paper":"/paper/peer-expertizing-domain-specific-tasks-with-a","slug":"peer-expertizing-domain-specific-tasks-with-a","title":"PEER: Expertizing Domain-Specific Tasks with a Multi-Agent Framework and Tuning Methods","date":"2024-07-09","arxiv_id":"2407.06985","n_code_links":1,"syntology":null},{"paper":null,"slug":"prompting-techniques-for-secure-code","title":"Prompting Techniques for Secure Code Generation: A Systematic Investigation","date":"2024-07-09","arxiv_id":"2407.07064","n_code_links":0,"syntology":null},{"paper":"/paper/source-code-summarization-in-the-era-of-large","slug":"source-code-summarization-in-the-era-of-large","title":"Source Code Summarization in the Era of Large Language Models","date":"2024-07-09","arxiv_id":"2407.07959","n_code_links":1,"syntology":null},{"paper":null,"slug":"using-large-language-models-for-generating","title":"Using Large Language Models for Generating Smart Contracts for Health Insurance from Textual Policies","date":"2024-07-09","arxiv_id":"2407.07019","n_code_links":0,"syntology":null},{"paper":"/paper/codeupdatearena-benchmarking-knowledge","slug":"codeupdatearena-benchmarking-knowledge","title":"CodeUpdateArena: Benchmarking Knowledge Editing on API Updates","date":"2024-07-08","arxiv_id":"2407.06249","n_code_links":1,"syntology":null},{"paper":null,"slug":"generative-debunking-of-climate","title":"Generative Debunking of Climate Misinformation","date":"2024-07-08","arxiv_id":"2407.05599","n_code_links":0,"syntology":null},{"paper":"/paper/inversecoder-unleashing-the-power-of","slug":"inversecoder-unleashing-the-power-of","title":"InverseCoder: Self-improving Instruction-Tuned Code LLMs with Inverse-Instruct","date":"2024-07-08","arxiv_id":"2407.05700","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wyt2000/InverseCoder"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/llm-based-open-domain-integrated-task-and","slug":"llm-based-open-domain-integrated-task-and","title":"Controllable and Reliable Knowledge-Intensive Task-Oriented Conversational Agents with Declarative Genie Worksheets","date":"2024-07-08","arxiv_id":"2407.05674","n_code_links":1,"syntology":null},{"paper":"/paper/meme-analysis-using-llm-based-contextual","slug":"meme-analysis-using-llm-based-contextual","title":"Meme Analysis using LLM-based Contextual Information and U-net Encapsulated Transformer","date":"2024-07-08","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"potential-of-multimodal-large-language-models","title":"Potential of Multimodal Large Language Models for Data Mining of Medical Images and Free-text Reports","date":"2024-07-08","arxiv_id":"2407.05758","n_code_links":0,"syntology":null},{"paper":null,"slug":"surprising-gender-biases-in-gpt","title":"Surprising gender biases in GPT","date":"2024-07-08","arxiv_id":"2407.06003","n_code_links":0,"syntology":null},{"paper":null,"slug":"t2vsafetybench-evaluating-the-safety-of-text","title":"T2VSafetyBench: Evaluating the Safety of Text-to-Video Generative Models","date":"2024-07-08","arxiv_id":"2407.05965","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-optimizing-and-evaluating-a-retrieval","title":"Towards Optimizing and Evaluating a Retrieval Augmented QA Chatbot using LLMs with Human in the Loop","date":"2024-07-08","arxiv_id":"2407.05925","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-computer-programming-education-with","title":"Enhancing Computer Programming Education with LLMs: A Study on Effective Prompt Engineering for Python Code Generation","date":"2024-07-07","arxiv_id":"2407.05437","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-model-as-an-assignment","title":"Large Language Model as an Assignment Evaluator: Insights, Feedback, and Challenges in a 1000+ Student Course","date":"2024-07-07","arxiv_id":"2407.05216","n_code_links":0,"syntology":null},{"paper":null,"slug":"mindecho-role-playing-language-agents-for-key","title":"MINDECHO: Role-Playing Language Agents for Key Opinion Leaders","date":"2024-07-07","arxiv_id":"2407.05305","n_code_links":0,"syntology":null},{"paper":null,"slug":"eva-score-evaluation-of-long-form","title":"EVA-Score: Evaluating Abstractive Long-form Summarization on Informativeness through Extraction and Validation","date":"2024-07-06","arxiv_id":"2407.04969","n_code_links":0,"syntology":null},{"paper":"/paper/how-do-you-know-that-teaching-generative","slug":"how-do-you-know-that-teaching-generative","title":"How do you know that? Teaching Generative Language Models to Reference Answers to Biomedical Questions","date":"2024-07-06","arxiv_id":"2407.05015","n_code_links":1,"syntology":null},{"paper":"/paper/solving-for-x-and-beyond-can-large-language","slug":"solving-for-x-and-beyond-can-large-language","title":"Solving for X and Beyond: Can Large Language Models Solve Complex Math Problems with More-Than-Two Unknowns?","date":"2024-07-06","arxiv_id":"2407.05134","n_code_links":1,"syntology":null},{"paper":null,"slug":"aligning-model-evaluations-with-human","title":"Aligning Model Evaluations with Human Preferences: Mitigating Token Count Bias in Language Model Assessments","date":"2024-07-05","arxiv_id":"2407.12847","n_code_links":0,"syntology":null},{"paper":"/paper/anah-v2-scaling-analytical-hallucination","slug":"anah-v2-scaling-analytical-hallucination","title":"ANAH-v2: Scaling Analytical Hallucination Annotation of Large Language Models","date":"2024-07-05","arxiv_id":"2407.04693","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":0,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["open-compass/anah"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/using-llms-to-label-medical-papers-according","slug":"using-llms-to-label-medical-papers-according","title":"Using LLMs to label medical papers according to the CIViC evidence model","date":"2024-07-05","arxiv_id":"2407.04466","n_code_links":1,"syntology":null},{"paper":null,"slug":"diverse-and-fine-grained-instruction","title":"Diverse and Fine-Grained Instruction-Following Ability Exploration with Synthetic Data","date":"2024-07-04","arxiv_id":"2407.03942","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-language-model-context-windows-a","slug":"evaluating-language-model-context-windows-a","title":"Evaluating Language Model Context Windows: A \"Working Memory\" Test and Inference-time Correction","date":"2024-07-04","arxiv_id":"2407.03651","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-4-vs-human-translators-a-comprehensive","title":"GPT-4 vs. Human Translators: A Comprehensive Evaluation of Translation Quality Across Languages, Domains, and Expertise Levels","date":"2024-07-04","arxiv_id":"2407.03658","n_code_links":0,"syntology":null},{"paper":null,"slug":"hera-high-efficiency-matrix-compression-via","title":"QET: Enhancing Quantized LLM Parameters and KV cache Compression through Element Substitution and Residual Clustering","date":"2024-07-04","arxiv_id":"2407.03637","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-benchmarking-of-llms-for-open-domain","title":"On the Benchmarking of LLMs for Open-Domain Dialogue Evaluation","date":"2024-07-04","arxiv_id":"2407.03841","n_code_links":0,"syntology":null},{"paper":null,"slug":"query-guided-self-supervised-summarization-of","title":"Query-Guided Self-Supervised Summarization of Nursing Notes","date":"2024-07-04","arxiv_id":"2407.04125","n_code_links":0,"syntology":null},{"paper":null,"slug":"solving-zebra-puzzles-using-constraint-guided","title":"Solving Zebra Puzzles Using Constraint-Guided Multi-Agent Systems","date":"2024-07-04","arxiv_id":"2407.03956","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-automating-text-annotation-a-case","title":"Towards Automating Text Annotation: A Case Study on Semantic Proximity Annotation using GPT-4","date":"2024-07-04","arxiv_id":"2407.04130","n_code_links":0,"syntology":null},{"paper":null,"slug":"lane-logic-alignment-of-non-tuning-large","title":"LANE: Logic Alignment of Non-tuning Large Language Models and Online Recommendation Systems for Explainable Reason Generation","date":"2024-07-03","arxiv_id":"2407.02833","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-as-evaluators-for-1","title":"Large Language Models as Evaluators for Scientific Synthesis","date":"2024-07-03","arxiv_id":"2407.02977","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-reduce-towards-improving","title":"Learning to Reduce: Towards Improving Performance of Large Language Models on Structured Data","date":"2024-07-03","arxiv_id":"2407.02750","n_code_links":0,"syntology":null},{"paper":"/paper/on-large-language-models-in-national-security","slug":"on-large-language-models-in-national-security","title":"On Large Language Models in National Security Applications","date":"2024-07-03","arxiv_id":"2407.03453","n_code_links":1,"syntology":null},{"paper":null,"slug":"semiollm-assessing-large-language-models-for","title":"SemioLLM: Assessing Large Language Models for Semiological Analysis in Epilepsy Research","date":"2024-07-03","arxiv_id":"2407.03004","n_code_links":0,"syntology":null},{"paper":"/paper/theoremllama-transforming-general-purpose","slug":"theoremllama-transforming-general-purpose","title":"TheoremLlama: Transforming General-Purpose LLMs into Lean4 Experts","date":"2024-07-03","arxiv_id":"2407.03203","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["RickySkywalker/TheoremLlama"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"assessing-the-code-clone-detection-capability","title":"Assessing the Code Clone Detection Capability of Large Language Models","date":"2024-07-02","arxiv_id":"2407.02402","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-numeric-awards-in-context-dueling","title":"Beyond Numeric Awards: In-Context Dueling Bandits with LLM Agents","date":"2024-07-02","arxiv_id":"2407.01887","n_code_links":0,"syntology":null},{"paper":null,"slug":"fake-news-detection-and-manipulation","title":"Fake News Detection and Manipulation Reasoning via Large Vision-Language Models","date":"2024-07-02","arxiv_id":"2407.02042","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-visual-storytelling-with-multimodal","title":"Improving Visual Storytelling with Multimodal Large Language Models","date":"2024-07-02","arxiv_id":"2407.02586","n_code_links":0,"syntology":null},{"paper":"/paper/integrate-the-essence-and-eliminate-the-dross","slug":"integrate-the-essence-and-eliminate-the-dross","title":"Integrate the Essence and Eliminate the Dross: Fine-Grained Self-Consistency for Free-Form Language Generation","date":"2024-07-02","arxiv_id":"2407.02056","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["WangXinglin/FSC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"llm-select-feature-selection-with-large","title":"LLM-Select: Feature Selection with Large Language Models","date":"2024-07-02","arxiv_id":"2407.02694","n_code_links":0,"syntology":null},{"paper":null,"slug":"open-foundation-models-for-azerbaijani","title":"Open foundation models for Azerbaijani language","date":"2024-07-02","arxiv_id":"2407.02337","n_code_links":0,"syntology":null},{"paper":"/paper/rankrag-unifying-context-ranking-with","slug":"rankrag-unifying-context-ranking-with","title":"RankRAG: Unifying Context Ranking with Retrieval-Augmented Generation in LLMs","date":"2024-07-02","arxiv_id":"2407.02485","n_code_links":0,"syntology":null},{"paper":"/paper/sop-unlock-the-power-of-social-facilitation","slug":"sop-unlock-the-power-of-social-facilitation","title":"SeqAR: Jailbreak LLMs with Sequential Auto-Generated Characters","date":"2024-07-02","arxiv_id":"2407.01902","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yang-yan-yang-yan/sop"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-art-of-saying-no-contextual-noncompliance","slug":"the-art-of-saying-no-contextual-noncompliance","title":"The Art of Saying No: Contextual Noncompliance in Language Models","date":"2024-07-02","arxiv_id":"2407.12043","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/deciphering-the-factors-influencing-the","slug":"deciphering-the-factors-influencing-the","title":"Deciphering the Factors Influencing the Efficacy of Chain-of-Thought: Probability, Memorization, and Noisy Reasoning","date":"2024-07-01","arxiv_id":"2407.01687","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["aksh555/deciphering_cot"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"image-to-text-logic-jailbreak-your","title":"Image-to-Text Logic Jailbreak: Your Imagination can Help You Do Anything","date":"2024-07-01","arxiv_id":"2407.02534","n_code_links":0,"syntology":null}],"record_sha256":"9da214b74f04446b92b6688ffe34dbedd43af49d007dce5d9e59cf3f623e9385","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}