{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/9","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":9,"pages_in_order":29,"rows_per_page":100,"rows":[801,900],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/8","next":"/method/gpt-4/papers/10","papers":[{"paper":null,"slug":"blockchain-enabled-accountability-in-data","title":"Blockchain-Enabled Accountability in Data Supply Chain: A Data Bill of Materials Approach","date":"2024-08-16","arxiv_id":"2408.08536","n_code_links":0,"syntology":null},{"paper":"/paper/can-large-language-models-improve-the","slug":"can-large-language-models-improve-the","title":"Can Large Language Models Improve the Adversarial Robustness of Graph Neural Networks?","date":"2024-08-16","arxiv_id":"2408.08685","n_code_links":1,"syntology":null},{"paper":null,"slug":"persona-is-a-double-edged-sword-enhancing-the","title":"Persona is a Double-edged Sword: Mitigating the Negative Impact of Role-playing Prompts in Zero-shot Reasoning Tasks","date":"2024-08-16","arxiv_id":"2408.08631","n_code_links":0,"syntology":null},{"paper":"/paper/see-what-llms-cannot-answer-a-self-challenge","slug":"see-what-llms-cannot-answer-a-self-challenge","title":"See What LLMs Cannot Answer: A Self-Challenge Framework for Uncovering LLM Weaknesses","date":"2024-08-16","arxiv_id":"2408.08978","n_code_links":1,"syntology":null},{"paper":"/paper/arablegaleval-a-multitask-benchmark-for","slug":"arablegaleval-a-multitask-benchmark-for","title":"ArabLegalEval: A Multitask Benchmark for Assessing Arabic Legal Knowledge in Large Language Models","date":"2024-08-15","arxiv_id":"2408.07983","n_code_links":1,"syntology":null},{"paper":null,"slug":"benchmarking-the-capabilities-of-large","title":"Benchmarking the Capabilities of Large Language Models in Transportation System Engineering: Accuracy, Consistency, and Reasoning Behaviors","date":"2024-08-15","arxiv_id":"2408.08302","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-the-validity-of-word-level","slug":"evaluating-the-validity-of-word-level","title":"Evaluating the Validity of Word-level Adversarial Attacks with Large Language Models","date":"2024-08-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/leveraging-web-crawled-data-for-high-quality","slug":"leveraging-web-crawled-data-for-high-quality","title":"Leveraging Web-Crawled Data for High-Quality Fine-Tuning","date":"2024-08-15","arxiv_id":"2408.08003","n_code_links":1,"syntology":null},{"paper":"/paper/mag-sql-multi-agent-generative-approach-with","slug":"mag-sql-multi-agent-generative-approach-with","title":"MAG-SQL: Multi-Agent Generative Approach with Soft Schema Linking and Iterative Sub-SQL Refinement for Text-to-SQL","date":"2024-08-15","arxiv_id":"2408.07930","n_code_links":1,"syntology":{"ran":21,"of":25,"n_ran_checked":19,"n_instrument":2,"unverified":4,"pointer_only":4,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 1 honoured, 3 violated, 15 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["LancelotXWX/MAG-SQL"],"state":"official (archive's flag): 21 ran","n_ran":21,"n_constructed":0,"n_ran_no_instrument_failure":19,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"polaris-open-ended-interactive-robotic","title":"Polaris: Open-ended Interactive Robotic Manipulation via Syn2Real Visual Grounding and Large Language Models","date":"2024-08-15","arxiv_id":"2408.07975","n_code_links":0,"syntology":null},{"paper":null,"slug":"codemirage-hallucinations-in-code-generated","title":"CodeMirage: Hallucinations in Code Generated by Large Language Models","date":"2024-08-14","arxiv_id":"2408.08333","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-perspective-on-large-language-models","title":"A Perspective on Large Language Models, Intelligent Machines, and Knowledge Acquisition","date":"2024-08-13","arxiv_id":"2408.06598","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-ai-for-automatic-topic-labelling","title":"Generative AI for automatic topic labelling","date":"2024-08-13","arxiv_id":"2408.07003","n_code_links":0,"syntology":null},{"paper":null,"slug":"harnessing-earnings-reports-for-stock","title":"Harnessing Earnings Reports for Stock Predictions: A QLoRA-Enhanced LLM Approach","date":"2024-08-13","arxiv_id":"2408.06634","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-language-models-for-emotion-and","title":"Leveraging Language Models for Emotion and Behavior Analysis in Education","date":"2024-08-13","arxiv_id":"2408.06874","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-advanced-llms-to-enhance-smaller-llms","title":"Using Advanced LLMs to Enhance Smaller LLMs: An Interpretable Knowledge Distillation Approach","date":"2024-08-13","arxiv_id":"2408.07238","n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-lingual-conversational-speech","title":"Cross-Lingual Conversational Speech Summarization with Large Language Models","date":"2024-08-12","arxiv_id":"2408.06484","n_code_links":0,"syntology":null},{"paper":null,"slug":"med42-v2-a-suite-of-clinical-llms","title":"Med42-v2: A Suite of Clinical LLMs","date":"2024-08-12","arxiv_id":"2408.06142","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-language-of-trauma-modeling-traumatic","title":"The Language of Trauma: Modeling Traumatic Event Descriptions Across Domains with Explainable AI","date":"2024-08-12","arxiv_id":"2408.05977","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-emulates-average-human-emotional","title":"GPT-4 Emulates Average-Human Emotional Cognition from a Third-Person Perspective","date":"2024-08-11","arxiv_id":"2408.13718","n_code_links":0,"syntology":null},{"paper":null,"slug":"chain-of-condition-construct-verify-and-solve","title":"Chain of Condition: Construct, Verify and Solve Conditions for Conditional Question Answering","date":"2024-08-10","arxiv_id":"2408.05442","n_code_links":0,"syntology":null},{"paper":null,"slug":"chatgpt-meets-iris-biometrics","title":"ChatGPT Meets Iris Biometrics","date":"2024-08-09","arxiv_id":"2408.04868","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-the-capability-of-large-language","title":"Evaluating the capability of large language models to personalize science texts for diverse middle-school-age learners","date":"2024-08-09","arxiv_id":"2408.05204","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-text-to-insight-leveraging-large","title":"From Text to Insight: Leveraging Large Language Models for Performance Evaluation in Management","date":"2024-08-09","arxiv_id":"2408.05328","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-and-thematic-analysis","title":"Large Language Models and Thematic Analysis: Human-AI Synergy in Researching Hate Speech on Social Media","date":"2024-08-09","arxiv_id":"2408.05126","n_code_links":0,"syntology":null},{"paper":"/paper/llmjudge-llms-for-relevance-judgments","slug":"llmjudge-llms-for-relevance-judgments","title":"LLMJudge: LLMs for Relevance Judgments","date":"2024-08-09","arxiv_id":"2408.08896","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-gpt-4-models-detect-misleading","title":"Can GPT-4 Models Detect Misleading Visualizations?","date":"2024-08-08","arxiv_id":"2408.12617","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-turn-context-jailbreak-attack-on-large","title":"Multi-Turn Context Jailbreak Attack on Large Language Models From First Principles","date":"2024-08-08","arxiv_id":"2408.04686","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-explainable-network-intrusion","title":"Towards Explainable Network Intrusion Detection using Large Language Models","date":"2024-08-08","arxiv_id":"2408.04342","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparison-of-llm-finetuning-methods","title":"A Comparison of LLM Finetuning Methods & Evaluation Metrics with Travel Chatbot Use Case","date":"2024-08-07","arxiv_id":"2408.03562","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-rule-based-insights-enhance-llms-for","title":"Can Rule-Based Insights Enhance LLMs for Radiology Report Classification? Introducing the RadPrompt Methodology","date":"2024-08-07","arxiv_id":"2408.04121","n_code_links":0,"syntology":null},{"paper":null,"slug":"could-chatgpt-get-an-engineering-degree","title":"Could ChatGPT get an Engineering Degree? Evaluating Higher Education Vulnerability to AI Assistants","date":"2024-08-07","arxiv_id":"2408.11841","n_code_links":0,"syntology":null},{"paper":null,"slug":"fmifood-multi-modal-contrastive-learning-for","title":"FMiFood: Multi-modal Contrastive Learning for Food Image Classification","date":"2024-08-07","arxiv_id":"2408.03922","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-02923","title":"Intermediate direct preference optimization","date":"2024-08-06","arxiv_id":"2408.02923","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-03062","title":"Analysis of Argument Structure Constructions in a Deep Recurrent Language Model","date":"2024-08-06","arxiv_id":"2408.03062","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-llms-serve-as-time-series-anomaly","title":"Can LLMs Serve As Time Series Anomaly Detectors?","date":"2024-08-06","arxiv_id":"2408.03475","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-aided-compilation-for-tensor-accelerators","title":"LLM-Aided Compilation for Tensor Accelerators","date":"2024-08-06","arxiv_id":"2408.03408","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-based-mofs-synthesis-condition-extraction","title":"LLM-based MOFs Synthesis Condition Extraction using Few-Shot Demonstrations","date":"2024-08-06","arxiv_id":"2408.04665","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-02213","title":"Is Large Language Model Good at Database Knob Tuning? A Comprehensive Experimental Evaluation","date":"2024-08-05","arxiv_id":"2408.02213","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-02237","title":"Do Large Language Models Speak All Languages Equally? A Comparative Study in Low-Resource Settings","date":"2024-08-05","arxiv_id":"2408.02237","n_code_links":0,"syntology":null},{"paper":"/paper/2408-02416","slug":"2408-02416","title":"Why Are My Prompts Leaked? Unraveling Prompt Extraction Threats in Customized Large Language Models","date":"2024-08-05","arxiv_id":"2408.02416","n_code_links":1,"syntology":null},{"paper":"/paper/seas-self-evolving-adversarial-safety","slug":"seas-self-evolving-adversarial-safety","title":"SEAS: Self-Evolving Adversarial Safety Optimization for Large Language Models","date":"2024-08-05","arxiv_id":"2408.02632","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-taught-evaluators","title":"Self-Taught Evaluators","date":"2024-08-05","arxiv_id":"2408.02666","n_code_links":0,"syntology":null},{"paper":"/paper/2408-01933","slug":"2408-01933","title":"DiReCT: Diagnostic Reasoning for Clinical Notes via Large Language Models","date":"2024-08-04","arxiv_id":"2408.01933","n_code_links":1,"syntology":{"ran":11,"of":13,"n_ran_checked":9,"n_instrument":2,"unverified":2,"pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["wbw520/direct"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/2408-02056","slug":"2408-02056","title":"MedSyn: LLM-based Synthetic Medical Text Generation Framework","date":"2024-08-04","arxiv_id":"2408.02056","n_code_links":1,"syntology":null},{"paper":null,"slug":"2408-01614","title":"Advancing Mental Health Pre-Screening: A New Custom GPT for Psychological Distress Assessment","date":"2024-08-03","arxiv_id":"2408.01614","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-01633","title":"Self-Emotion Blended Dialogue Generation in Social Simulation Agents","date":"2024-08-03","arxiv_id":"2408.01633","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-01655","title":"Stimulating Imagination: Towards General-purpose Object Rearrangement","date":"2024-08-03","arxiv_id":"2408.01655","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-01723","title":"A Novel Evaluation Framework for Image2Text Generation","date":"2024-08-03","arxiv_id":"2408.01723","n_code_links":0,"syntology":null},{"paper":"/paper/2408-01869","slug":"2408-01869","title":"MALADE: Orchestration of LLM-powered Agents with Retrieval Augmented Generation for Pharmacovigilance","date":"2024-08-03","arxiv_id":"2408.01869","n_code_links":1,"syntology":null},{"paper":null,"slug":"2408-01055","title":"LLM as Runtime Error Handler: A Promising Pathway to Adaptive Self-Healing of Software Systems","date":"2024-08-02","arxiv_id":"2408.01055","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-01214","title":"High-Throughput Phenotyping of Clinical Text Using Large Language Models","date":"2024-08-02","arxiv_id":"2408.01214","n_code_links":0,"syntology":null},{"paper":"/paper/2408-00764","slug":"2408-00764","title":"AgentGen: Enhancing Planning Abilities for Large Language Model based Agent via Environment and Task Generation","date":"2024-08-01","arxiv_id":"2408.00764","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["lazychih114/AgentGen-Reproduction"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"2408-00914","title":"Granting GPT-4 License and Opportunity: Enhancing Accuracy and Confidence Estimation for Few-Shot Event Detection","date":"2024-08-01","arxiv_id":"2408.00914","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-querying-over-relational-databases-and","title":"Hybrid Querying Over Relational Databases and Large Language Models","date":"2024-08-01","arxiv_id":"2408.00884","n_code_links":0,"syntology":null},{"paper":null,"slug":"2407-21276","title":"Multi-Level Querying using A Knowledge Pyramid","date":"2024-07-31","arxiv_id":"2407.21276","n_code_links":0,"syntology":null},{"paper":null,"slug":"2407-21512","title":"Interpreting and learning voice commands with a Large Language Model for a robot system","date":"2024-07-31","arxiv_id":"2407.21512","n_code_links":0,"syntology":null},{"paper":null,"slug":"2407-21531","title":"Can LLMs \"Reason\" in Music? An Evaluation of LLMs' Capability of Music Understanding and Generation","date":"2024-07-31","arxiv_id":"2407.21531","n_code_links":0,"syntology":null},{"paper":"/paper/the-llama-3-herd-of-models","slug":"the-llama-3-herd-of-models","title":"The Llama 3 Herd of Models","date":"2024-07-31","arxiv_id":"2407.21783","n_code_links":5,"syntology":{"ran":9,"of":9,"n_ran_checked":8,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"breaking-agents-compromising-autonomous-llm","title":"Breaking Agents: Compromising Autonomous LLM Agents Through Malfunction Amplification","date":"2024-07-30","arxiv_id":"2407.20859","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-agricultural-machinery-management","title":"Enhancing Agricultural Machinery Management through Advanced LLM Integration","date":"2024-07-30","arxiv_id":"2407.20588","n_code_links":0,"syntology":null},{"paper":null,"slug":"mimicking-the-mavens-agent-based-opinion","title":"Mimicking the Mavens: Agent-based Opinion Synthesis and Emotion Prediction for Social Media Influencers","date":"2024-07-30","arxiv_id":"2407.20668","n_code_links":0,"syntology":null},{"paper":"/paper/synthvlm-high-efficiency-and-high-quality","slug":"synthvlm-high-efficiency-and-high-quality","title":"SynthVLM: High-Efficiency and High-Quality Synthetic Data for Vision Language Models","date":"2024-07-30","arxiv_id":"2407.20756","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-retrieval-augmented-language-model","title":"Improving Retrieval Augmented Language Model with Self-Reasoning","date":"2024-07-29","arxiv_id":"2407.19813","n_code_links":0,"syntology":null},{"paper":null,"slug":"legal-minds-algorithmic-decisions-how-llms","title":"Legal Minds, Algorithmic Decisions: How LLMs Apply Constitutional Principles in Complex Scenarios","date":"2024-07-29","arxiv_id":"2407.19760","n_code_links":0,"syntology":null},{"paper":null,"slug":"revolutionizing-urban-safety-perception","title":"Revolutionizing Urban Safety Perception Assessments: Integrating Multimodal Large Language Models with Street View Images","date":"2024-07-29","arxiv_id":"2407.19719","n_code_links":0,"syntology":null},{"paper":null,"slug":"sentiment-analysis-of-lithuanian-online","title":"Sentiment Analysis of Lithuanian Online Reviews Using Large Language Models","date":"2024-07-29","arxiv_id":"2407.19914","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-if-red-can-talk-dynamic-dialogue","title":"What if Red Can Talk? Dynamic Dialogue Generation Using Large Language Models","date":"2024-07-29","arxiv_id":"2407.20382","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrating-large-language-models-into-a-tri","title":"Integrating Large Language Models into a Tri-Modal Architecture for Automated Depression Classification on the DAIC-WOZ","date":"2024-07-27","arxiv_id":"2407.19340","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-deciphering-fedspeak-quantifying-dissent","slug":"gpt-deciphering-fedspeak-quantifying-dissent","title":"GPT Deciphering Fedspeak: Quantifying Dissent Among Hawks and Doves","date":"2024-07-26","arxiv_id":"2407.19110","n_code_links":1,"syntology":null},{"paper":"/paper/officebench-benchmarking-language-agents","slug":"officebench-benchmarking-language-agents","title":"OfficeBench: Benchmarking Language Agents across Multiple Applications for Office Automation","date":"2024-07-26","arxiv_id":"2407.19056","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["zlwang-cs/OfficeBench"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"tagify-llm-powered-tagging-interface-for","title":"TAGIFY: LLM-powered Tagging Interface for Improved Data Findability on OGD portals","date":"2024-07-26","arxiv_id":"2407.18764","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-gpt-4-to-guide-causal-machine-learning","title":"Using GPT-4 to guide causal machine learning","date":"2024-07-26","arxiv_id":"2407.18607","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-large-language-models-for-the","title":"Using Large Language Models for the Interpretation of Building Regulations","date":"2024-07-26","arxiv_id":"2407.21060","n_code_links":0,"syntology":null},{"paper":"/paper/cost-effective-instruction-learning-for","slug":"cost-effective-instruction-learning-for","title":"Cost-effective Instruction Learning for Pathology Vision and Language Analysis","date":"2024-07-25","arxiv_id":"2407.17734","n_code_links":1,"syntology":null},{"paper":null,"slug":"is-the-digital-forensics-and-incident","title":"Is the Digital Forensics and Incident Response Pipeline Ready for Text-Based Threats in LLM Era?","date":"2024-07-25","arxiv_id":"2407.17870","n_code_links":0,"syntology":null},{"paper":"/paper/personagym-evaluating-persona-agents-and-llms","slug":"personagym-evaluating-persona-agents-and-llms","title":"PersonaGym: Evaluating Persona Agents and LLMs","date":"2024-07-25","arxiv_id":"2407.18416","n_code_links":1,"syntology":null},{"paper":"/paper/self-training-with-direct-preference","slug":"self-training-with-direct-preference","title":"Self-Training with Direct Preference Optimization Improves Chain-of-Thought Reasoning","date":"2024-07-25","arxiv_id":"2407.18248","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tianduowang/dpo-st"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"trust-or-escalate-llm-judges-with-provable","title":"Trust or Escalate: LLM Judges with Provable Guarantees for Human Agreement","date":"2024-07-25","arxiv_id":"2407.18370","n_code_links":0,"syntology":null},{"paper":null,"slug":"bailicai-a-domain-optimized-retrieval","title":"Bailicai: A Domain-Optimized Retrieval-Augmented Generation Framework for Medical Applications","date":"2024-07-24","arxiv_id":"2407.21055","n_code_links":0,"syntology":null},{"paper":"/paper/i-could-ve-asked-that-reformulating","slug":"i-could-ve-asked-that-reformulating","title":"I Could've Asked That: Reformulating Unanswerable Questions","date":"2024-07-24","arxiv_id":"2407.17469","n_code_links":1,"syntology":null},{"paper":null,"slug":"testing-large-language-models-on-driving","title":"Testing Large Language Models on Driving Theory Knowledge and Skills for Connected Autonomous Vehicles","date":"2024-07-24","arxiv_id":"2407.17211","n_code_links":0,"syntology":null},{"paper":null,"slug":"artificial-intelligence-in-extracting","title":"Artificial Intelligence in Extracting Diagnostic Data from Dental Records","date":"2024-07-23","arxiv_id":"2407.21050","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-llms-know-when-to-not-answer-investigating","title":"Do LLMs Know When to NOT Answer? Investigating Abstention Abilities of Large Language Models","date":"2024-07-23","arxiv_id":"2407.16221","n_code_links":0,"syntology":null},{"paper":null,"slug":"lawluo-a-chinese-law-firm-co-run-by-llm","title":"LawLuo: A Multi-Agent Collaborative Framework for Multi-Round Chinese Legal Consultation","date":"2024-07-23","arxiv_id":"2407.16252","n_code_links":0,"syntology":null},{"paper":"/paper/lawma-the-power-of-specialization-for-legal","slug":"lawma-the-power-of-specialization-for-legal","title":"Lawma: The Power of Specialization for Legal Tasks","date":"2024-07-23","arxiv_id":"2407.16615","n_code_links":0,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/origen-enhancing-rtl-code-generation-with","slug":"origen-enhancing-rtl-code-generation-with","title":"OriGen:Enhancing RTL Code Generation with Code-to-Code Augmentation and Self-Reflection","date":"2024-07-23","arxiv_id":"2407.16237","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["pku-liang/origen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/patched-rtc-evaluating-llms-for-diverse","slug":"patched-rtc-evaluating-llms-for-diverse","title":"Patched RTC: evaluating LLMs for diverse software development tasks","date":"2024-07-23","arxiv_id":"2407.16557","n_code_links":1,"syntology":null},{"paper":null,"slug":"redagent-red-teaming-large-language-models","title":"RedAgent: Red Teaming Large Language Models with Context-aware Autonomous Language Agent","date":"2024-07-23","arxiv_id":"2407.16667","n_code_links":0,"syntology":null},{"paper":null,"slug":"retrieval-augmented-generation-or-long","title":"Retrieval Augmented Generation or Long-Context LLMs? A Comprehensive Study and Hybrid Approach","date":"2024-07-23","arxiv_id":"2407.16833","n_code_links":0,"syntology":null},{"paper":"/paper/robust-privacy-amidst-innovation-with-large","slug":"robust-privacy-amidst-innovation-with-large","title":"Robust Privacy Amidst Innovation with Large Language Models Through a Critical Assessment of the Risks","date":"2024-07-23","arxiv_id":"2407.16166","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-gpt-4-learn-to-analyze-moves-in-research","title":"Can GPT-4 learn to analyse moves in research article abstracts?","date":"2024-07-22","arxiv_id":"2407.15612","n_code_links":0,"syntology":null},{"paper":"/paper/dissecting-multiplication-in-transformers","slug":"dissecting-multiplication-in-transformers","title":"Dissecting Multiplication in Transformers: Insights into LLMs","date":"2024-07-22","arxiv_id":"2407.15360","n_code_links":1,"syntology":null},{"paper":null,"slug":"imposter-ai-adversarial-attacks-with-hidden","title":"Imposter.AI: Adversarial Attacks with Hidden Intentions towards Aligned Large Language Models","date":"2024-07-22","arxiv_id":"2407.15399","n_code_links":0,"syntology":null},{"paper":null,"slug":"morse-bridging-the-gap-in-cybersecurity","title":"MoRSE: Bridging the Gap in Cybersecurity Expertise with Retrieval Augmented Generation","date":"2024-07-22","arxiv_id":"2407.15748","n_code_links":0,"syntology":null},{"paper":"/paper/radiorag-factual-large-language-models-for","slug":"radiorag-factual-large-language-models-for","title":"RadioRAG: Factual large language models for enhanced diagnostics in radiology using online retrieval augmented generation","date":"2024-07-22","arxiv_id":"2407.15621","n_code_links":1,"syntology":null},{"paper":null,"slug":"unlocking-the-potential-benchmarking-large","title":"Unlocking the Potential: Benchmarking Large Language Models in Water Engineering and Research","date":"2024-07-22","arxiv_id":"2407.21045","n_code_links":0,"syntology":null},{"paper":null,"slug":"arondight-red-teaming-large-vision-language","title":"Arondight: Red Teaming Large Vision Language Models with Auto-generated Multi-modal Jailbreak Prompts","date":"2024-07-21","arxiv_id":"2407.15050","n_code_links":0,"syntology":null},{"paper":"/paper/toward-adaptive-reasoning-in-large-language","slug":"toward-adaptive-reasoning-in-large-language","title":"Toward Adaptive Reasoning in Large Language Models with Thought Rollback","date":"2024-07-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-context-aware-preference-modeling","title":"Improving Context-Aware Preference Modeling for Language Models","date":"2024-07-20","arxiv_id":"2407.14916","n_code_links":0,"syntology":null}],"record_sha256":"ba60fcb4f14afa2c32bda4fdf6f9d0fa48169f5affff7180880c445a2cf3e845","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}