{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/7","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":7,"pages_in_order":29,"rows_per_page":100,"rows":[601,700],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/6","next":"/method/gpt-4/papers/8","papers":[{"paper":null,"slug":"rescriber-smaller-llm-powered-user-led-data","title":"Rescriber: Smaller-LLM-Powered User-Led Data Minimization for LLM-Based Chatbots","date":"2024-10-10","arxiv_id":"2410.11876","n_code_links":0,"syntology":null},{"paper":"/paper/teaching-inspired-integrated-prompting","slug":"teaching-inspired-integrated-prompting","title":"Teaching-Inspired Integrated Prompting Framework: A Novel Approach for Enhancing Reasoning in Large Language Models","date":"2024-10-10","arxiv_id":"2410.08068","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sallytan13/teaching-inspired-prompting"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"think-beyond-size-dynamic-prompting-for-more","title":"Think Beyond Size: Adaptive Prompting for More Effective Reasoning","date":"2024-10-10","arxiv_id":"2410.08130","n_code_links":0,"syntology":null},{"paper":"/paper/thought2text-text-generation-from-eeg-signal","slug":"thought2text-text-generation-from-eeg-signal","title":"Thought2Text: Text Generation from EEG Signal using Large Language Models (LLMs)","date":"2024-10-10","arxiv_id":"2410.07507","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["abhijitmishra/Thought2Text"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/vibecheck-discover-and-quantify-qualitative","slug":"vibecheck-discover-and-quantify-qualitative","title":"VibeCheck: Discover and Quantify Qualitative Differences in Large Language Models","date":"2024-10-10","arxiv_id":"2410.12851","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":10,"n_instrument":0,"unverified":2,"pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["lisadunlap/vibecheck"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"autofeedback-an-llm-based-framework-for","title":"AutoFeedback: An LLM-based Framework for Efficient and Accurate API Request Generation","date":"2024-10-09","arxiv_id":"2410.06943","n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-bias-and-enhancing-diagnostic","title":"Detecting Bias and Enhancing Diagnostic Accuracy in Large Language Models for Healthcare","date":"2024-10-09","arxiv_id":"2410.06566","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-great-minds-think-alike-investigating","title":"Do great minds think alike? Investigating Human-AI Complementarity in Question Answering with CAIMIRA","date":"2024-10-09","arxiv_id":"2410.06524","n_code_links":0,"syntology":null},{"paper":"/paper/eta-evaluating-then-aligning-safety-of-vision","slug":"eta-evaluating-then-aligning-safety-of-vision","title":"ETA: Evaluating Then Aligning Safety of Vision Language Models at Inference Time","date":"2024-10-09","arxiv_id":"2410.06625","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":6,"n_instrument":3,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["dripnowhy/eta"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-data-efficiency-via-curating-llm","title":"Improving Data Efficiency via Curating LLM-Driven Rating Systems","date":"2024-10-09","arxiv_id":"2410.10877","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-cost-efficiency-of-llm","title":"Investigating Cost-Efficiency of LLM-Generated Training Data for Conversational Semantic Frame Analysis","date":"2024-10-09","arxiv_id":"2410.06550","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-self-correction-with-decrim-decompose","title":"LLM Self-Correction with DeCRIM: Decompose, Critique, and Refine for Enhanced Following of Instructions with Multiple Constraints","date":"2024-10-09","arxiv_id":"2410.06458","n_code_links":0,"syntology":null},{"paper":"/paper/textlap-customizing-language-models-for-text","slug":"textlap-customizing-language-models-for-text","title":"TextLap: Customizing Language Models for Text-to-Layout Planning","date":"2024-10-09","arxiv_id":"2410.12844","n_code_links":1,"syntology":null},{"paper":"/paper/turingq-benchmarking-ai-comprehension-in","slug":"turingq-benchmarking-ai-comprehension-in","title":"TuringQ: Benchmarking AI Comprehension in Theory of Computation","date":"2024-10-09","arxiv_id":"2410.06547","n_code_links":1,"syntology":null},{"paper":null,"slug":"application-of-notebooklm-a-large-language","title":"Application of NotebookLM, a Large Language Model with Retrieval-Augmented Generation, for Lung Cancer Staging","date":"2024-10-08","arxiv_id":"2410.10869","n_code_links":0,"syntology":null},{"paper":null,"slug":"listen-to-the-patient-enhancing-medical","title":"Listening to Patients: A Framework of Detecting and Mitigating Patient Misreport for Medical Dialogue Generation","date":"2024-10-08","arxiv_id":"2410.06094","n_code_links":0,"syntology":null},{"paper":"/paper/sc-bench-a-large-scale-dataset-for-smart","slug":"sc-bench-a-large-scale-dataset-for-smart","title":"SC-Bench: A Large-Scale Dataset for Smart Contract Auditing","date":"2024-10-08","arxiv_id":"2410.06176","n_code_links":1,"syntology":null},{"paper":null,"slug":"contrastive-learning-to-improve-retrieval-for","title":"Contrastive Learning to Improve Retrieval for Real-world Fact Checking","date":"2024-10-07","arxiv_id":"2410.04657","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-cad-code-with-vision-language","title":"Generating CAD Code with Vision-Language Models for 3D Designs","date":"2024-10-07","arxiv_id":"2410.05340","n_code_links":0,"syntology":null},{"paper":null,"slug":"diagnosing-robotics-systems-issues-with-large","title":"Diagnosing Robotics Systems Issues with Large Language Models","date":"2024-10-06","arxiv_id":"2410.09084","n_code_links":0,"syntology":null},{"paper":"/paper/mindscope-exploring-cognitive-biases-in-large","slug":"mindscope-exploring-cognitive-biases-in-large","title":"MindScope: Exploring cognitive biases in large language models through Multi-Agent Systems","date":"2024-10-06","arxiv_id":"2410.04452","n_code_links":1,"syntology":null},{"paper":null,"slug":"protocollm-automatic-evaluation-framework-of","title":"ProtocoLLM: Automatic Evaluation Framework of LLMs on Domain-Specific Scientific Protocol Formulation Tasks","date":"2024-10-06","arxiv_id":"2410.04601","n_code_links":0,"syntology":null},{"paper":"/paper/econ-on-the-detection-and-resolution-of","slug":"econ-on-the-detection-and-resolution-of","title":"ECon: On the Detection and Resolution of Evidence Conflicts","date":"2024-10-05","arxiv_id":"2410.04068","n_code_links":1,"syntology":null},{"paper":"/paper/take-it-easy-label-adaptive-self","slug":"take-it-easy-label-adaptive-self","title":"Take It Easy: Label-Adaptive Self-Rationalization for Fact Verification and Explanation Generation","date":"2024-10-05","arxiv_id":"2410.04002","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jingyng/label-adaptive-self-rationalization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"audio-agent-leveraging-llms-for-audio","title":"Audio-Agent: Leveraging LLMs For Audio Generation, Editing and Composition","date":"2024-10-04","arxiv_id":"2410.03335","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-film-subtitles-is-youtube-the-best","slug":"beyond-film-subtitles-is-youtube-the-best","title":"Beyond Film Subtitles: Is YouTube the Best Approximation of Spoken Vocabulary?","date":"2024-10-04","arxiv_id":"2410.03240","n_code_links":1,"syntology":null},{"paper":null,"slug":"detecting-machine-generated-long-form-content","title":"Detecting Machine-Generated Long-Form Content with Latent-Space Variables","date":"2024-10-04","arxiv_id":"2410.03856","n_code_links":0,"syntology":null},{"paper":null,"slug":"sag-style-aligned-article-generation-via","title":"SAG: Style-Aligned Article Generation via Model Collaboration","date":"2024-10-04","arxiv_id":"2410.03137","n_code_links":0,"syntology":null},{"paper":null,"slug":"still-not-quite-there-evaluating-large","title":"Still Not Quite There! Evaluating Large Language Models for Comorbid Mental Health Diagnosis","date":"2024-10-04","arxiv_id":"2410.03908","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-linguistically-aware-and-language","title":"Towards Linguistically-Aware and Language-Independent Tokenization for Large Language Models (LLMs)","date":"2024-10-04","arxiv_id":"2410.03568","n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-inference-time-compute-llms-can","title":"Adaptive Inference-Time Compute: LLMs Can Predict if They Can Do Better, Even Mid-Generation","date":"2024-10-03","arxiv_id":"2410.02725","n_code_links":0,"syntology":null},{"paper":"/paper/can-llms-reliably-simulate-human-learner","slug":"can-llms-reliably-simulate-human-learner","title":"Can LLMs Reliably Simulate Human Learner Actions? A Simulation Authoring Framework for Open-Ended Learning Environments","date":"2024-10-03","arxiv_id":"2410.02110","n_code_links":1,"syntology":null},{"paper":"/paper/cax-cellular-automata-accelerated-in-jax","slug":"cax-cellular-automata-accelerated-in-jax","title":"CAX: Cellular Automata Accelerated in JAX","date":"2024-10-03","arxiv_id":"2410.02651","n_code_links":1,"syntology":null},{"paper":null,"slug":"coal-mining-question-answering-with-llms","title":"Coal Mining Question Answering with LLMs","date":"2024-10-03","arxiv_id":"2410.02959","n_code_links":0,"syntology":null},{"paper":null,"slug":"defining-knowledge-bridging-epistemology-and","title":"Defining Knowledge: Bridging Epistemology and Large Language Models","date":"2024-10-03","arxiv_id":"2410.02499","n_code_links":0,"syntology":null},{"paper":null,"slug":"grounding-large-language-models-in-embodied","title":"Grounding Large Language Models In Embodied Environment With Imperfect World Models","date":"2024-10-03","arxiv_id":"2410.02742","n_code_links":0,"syntology":null},{"paper":null,"slug":"iot-llm-enhancing-real-world-iot-task","title":"IoT-LLM: Enhancing Real-World IoT Task Reasoning with Large Language Models","date":"2024-10-03","arxiv_id":"2410.02429","n_code_links":0,"syntology":null},{"paper":"/paper/training-language-models-on-synthetic-edit","slug":"training-language-models-on-synthetic-edit","title":"Training Language Models on Synthetic Edit Sequences Improves Code Synthesis","date":"2024-10-03","arxiv_id":"2410.02749","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["upiterbarg/lintseq"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"ahp-powered-llm-reasoning-for-multi-criteria","title":"AHP-Powered LLM Reasoning for Multi-Criteria Evaluation of Open-Ended Responses","date":"2024-10-02","arxiv_id":"2410.01246","n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-red-teaming-with-goat-the","title":"Automated Red Teaming with GOAT: the Generative Offensive Agent Tester","date":"2024-10-02","arxiv_id":"2410.01606","n_code_links":0,"syntology":null},{"paper":null,"slug":"et-plan-bench-embodied-task-level-planning","title":"ET-Plan-Bench: Embodied Task-level Planning Benchmark Towards Spatial-Temporal Cognition with Foundation Models","date":"2024-10-02","arxiv_id":"2410.14682","n_code_links":0,"syntology":null},{"paper":"/paper/marple-a-benchmark-for-long-horizon-inference","slug":"marple-a-benchmark-for-long-horizon-inference","title":"MARPLE: A Benchmark for Long-Horizon Inference","date":"2024-10-02","arxiv_id":"2410.01926","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["marple-benchmark/marple"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mind-scramble-unveiling-large-language-model","slug":"mind-scramble-unveiling-large-language-model","title":"Mind Scramble: Unveiling Large Language Model Psychology Via Typoglycemia","date":"2024-10-02","arxiv_id":"2410.01677","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-adaptation-of-unlimiformer-for-decoder","title":"On The Adaptation of Unlimiformer for Decoder-Only Transformers","date":"2024-10-02","arxiv_id":"2410.01637","n_code_links":0,"syntology":null},{"paper":"/paper/adversarial-suffixes-may-be-features-too","slug":"adversarial-suffixes-may-be-features-too","title":"Unleashing the Unseen: Harnessing Benign Datasets for Jailbreaking Large Language Models","date":"2024-10-01","arxiv_id":"2410.00451","n_code_links":1,"syntology":null},{"paper":"/paper/creative-and-context-aware-translation-of","slug":"creative-and-context-aware-translation-of","title":"Creative and Context-Aware Translation of East Asian Idioms with GPT-4","date":"2024-10-01","arxiv_id":"2410.00988","n_code_links":1,"syntology":null},{"paper":null,"slug":"decoding-hate-exploring-language-models","title":"Decoding Hate: Exploring Language Models' Reactions to Hate Speech","date":"2024-10-01","arxiv_id":"2410.00775","n_code_links":0,"syntology":null},{"paper":null,"slug":"insight-a-multi-modal-diagnostic-pipeline","title":"Insight: A Multi-Modal Diagnostic Pipeline using LLMs for Ocular Surface Disease Diagnosis","date":"2024-10-01","arxiv_id":"2410.00292","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-enhanced-model-for-eye-leme-an-open","title":"Language Enhanced Model for Eye (LEME): An Open-Source Ophthalmology-Specific Large Language Model","date":"2024-10-01","arxiv_id":"2410.03740","n_code_links":0,"syntology":null},{"paper":"/paper/rationalyst-pre-training-process-supervision","slug":"rationalyst-pre-training-process-supervision","title":"RATIONALYST: Pre-training Process-Supervision for Improving Reasoning","date":"2024-10-01","arxiv_id":"2410.01044","n_code_links":1,"syntology":null},{"paper":null,"slug":"ace-all-round-creator-and-editor-following","title":"ACE: All-round Creator and Editor Following Instructions via Diffusion Transformer","date":"2024-09-30","arxiv_id":"2410.00086","n_code_links":0,"syntology":null},{"paper":"/paper/climb-an-ai-enabled-partner-for-clinical","slug":"climb-an-ai-enabled-partner-for-clinical","title":"CliMB: An AI-enabled Partner for Clinical Predictive Modeling","date":"2024-09-30","arxiv_id":"2410.03736","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["vanderschaarlab/climb"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-large-uni-and-multi-modal-models-for","title":"Exploring Social Media Image Categorization Using Large Models with Different Adaptation Methods: A Case Study on Cultural Nature's Contributions to People","date":"2024-09-30","arxiv_id":"2410.00275","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-planning-abilities-of-openai-s-o1","slug":"on-the-planning-abilities-of-openai-s-o1","title":"On The Planning Abilities of OpenAI's o1 Models: Feasibility, Optimality, and Generalizability","date":"2024-09-30","arxiv_id":"2409.19924","n_code_links":2,"syntology":null},{"paper":null,"slug":"can-models-learn-skill-composition-from","title":"Can Models Learn Skill Composition from Examples?","date":"2024-09-29","arxiv_id":"2409.19808","n_code_links":0,"syntology":null},{"paper":null,"slug":"gentel-safe-a-unified-benchmark-and-shielding","title":"GenTel-Safe: A Unified Benchmark and Shielding Framework for Defending Against Prompt Injection Attacks","date":"2024-09-29","arxiv_id":"2409.19521","n_code_links":0,"syntology":null},{"paper":null,"slug":"medhalu-hallucinations-in-responses-to","title":"MedHalu: Hallucinations in Responses to Healthcare Queries by Large Language Models","date":"2024-09-29","arxiv_id":"2409.19492","n_code_links":0,"syntology":null},{"paper":null,"slug":"see-then-tell-enhancing-key-information","title":"See then Tell: Enhancing Key Information Extraction with Vision Grounding","date":"2024-09-29","arxiv_id":"2409.19573","n_code_links":0,"syntology":null},{"paper":"/paper/towards-ai-assisted-protocol-analysis-in","slug":"towards-ai-assisted-protocol-analysis-in","title":"Towards AI-Assisted Protocol Analysis in Design Research: Automating Question Labelling with GPT-4 According to Eris’ (2004) Taxonomy","date":"2024-09-29","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/lml-language-model-learning-a-dataset-for","slug":"lml-language-model-learning-a-dataset-for","title":"LML-DAP: Language Model Learning a Dataset for Data-Augmented Prediction","date":"2024-09-27","arxiv_id":"2409.18957","n_code_links":1,"syntology":null},{"paper":null,"slug":"not-the-silver-bullet-llm-enhanced","title":"Not the Silver Bullet: LLM-enhanced Programming Error Messages are Ineffective in Practice","date":"2024-09-27","arxiv_id":"2409.18661","n_code_links":0,"syntology":null},{"paper":null,"slug":"open-nav-exploring-zero-shot-vision-and","title":"Open-Nav: Exploring Zero-Shot Vision-and-Language Navigation in Continuous Environment with Open-Source LLMs","date":"2024-09-27","arxiv_id":"2409.18794","n_code_links":0,"syntology":null},{"paper":null,"slug":"dare-diverse-visual-question-answering-with","title":"DARE: Diverse Visual Question Answering with Robustness Evaluation","date":"2024-09-26","arxiv_id":"2409.18023","n_code_links":0,"syntology":null},{"paper":null,"slug":"just-say-what-you-want-only-prompting-self","title":"Just Say What You Want: Only-prompting Self-rewarding Online Preference Optimization","date":"2024-09-26","arxiv_id":"2409.17534","n_code_links":0,"syntology":null},{"paper":null,"slug":"predicting-anchored-text-from-translation","title":"Predicting Anchored Text from Translation Memories for Machine Translation Using Deep Learning Methods","date":"2024-09-26","arxiv_id":"2409.17939","n_code_links":0,"syntology":null},{"paper":"/paper/retrospective-comparative-analysis-of","slug":"retrospective-comparative-analysis-of","title":"Retrospective Comparative Analysis of Prostate Cancer In-Basket Messages: Responses from Closed-Domain LLM vs. Clinical Teams","date":"2024-09-26","arxiv_id":"2409.18290","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-application-of-gpt-4-in-grading-design","title":"The application of GPT-4 in grading design university students' assignment and providing feedback: An exploratory study","date":"2024-09-26","arxiv_id":"2409.17698","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-turing-test-can-gpt-4-sway-experts","title":"Beyond Turing Test: Can GPT-4 Sway Experts' Decisions?","date":"2024-09-25","arxiv_id":"2409.16710","n_code_links":0,"syntology":null},{"paper":"/paper/codeinsight-a-curated-dataset-of-practical","slug":"codeinsight-a-curated-dataset-of-practical","title":"CodeInsight: A Curated Dataset of Practical Coding Solutions from Stack Overflow","date":"2024-09-25","arxiv_id":"2409.16819","n_code_links":1,"syntology":null},{"paper":"/paper/post-hoc-reward-calibration-a-case-study-on","slug":"post-hoc-reward-calibration-a-case-study-on","title":"Post-hoc Reward Calibration: A Case Study on Length Bias","date":"2024-09-25","arxiv_id":"2409.17407","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zeroyuhuang/reward-calibration"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-comprehensive-evaluation-of-large-language-3","title":"A Comprehensive Evaluation of Large Language Models on Mental Illnesses","date":"2024-09-24","arxiv_id":"2409.15687","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-can-be-cognitively-biased-an-exploratory","title":"AI Can Be Cognitively Biased: An Exploratory Study on Threshold Priming in LLM-Based Batch Relevance Assessment","date":"2024-09-24","arxiv_id":"2409.16022","n_code_links":0,"syntology":null},{"paper":null,"slug":"task-oriented-prompt-enhancement-via-script","title":"Task-oriented Prompt Enhancement via Script Generation","date":"2024-09-24","arxiv_id":"2409.16418","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-preliminary-study-of-o1-in-medicine-are-we","title":"A Preliminary Study of o1 in Medicine: Are We Closer to an AI Doctor?","date":"2024-09-23","arxiv_id":"2409.15277","n_code_links":0,"syntology":null},{"paper":"/paper/effective-and-evasive-fuzz-testing-driven","slug":"effective-and-evasive-fuzz-testing-driven","title":"PAPILLON: Efficient and Stealthy Fuzz Testing-Powered Jailbreaks for LLMs","date":"2024-09-23","arxiv_id":"2409.14866","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aaFrostnova/Papillon"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/memeclip-leveraging-clip-representations-for","slug":"memeclip-leveraging-clip-representations-for","title":"MemeCLIP: Leveraging CLIP Representations for Multimodal Meme Classification","date":"2024-09-23","arxiv_id":"2409.14703","n_code_links":1,"syntology":null},{"paper":"/paper/pallm-evaluating-and-enhancing-palliative","slug":"pallm-evaluating-and-enhancing-palliative","title":"PALLM: Evaluating and Enhancing PALLiative Care Conversations with Large Language Models","date":"2024-09-23","arxiv_id":"2409.15188","n_code_links":1,"syntology":null},{"paper":null,"slug":"beyond-words-evaluating-large-language-models","title":"Beyond Words: Evaluating Large Language Models in Transportation Planning","date":"2024-09-22","arxiv_id":"2409.14516","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-llm-based-autonomous-driving-agents","title":"Enhancing LLM-based Autonomous Driving Agents to Mitigate Perception Attacks","date":"2024-09-22","arxiv_id":"2409.14488","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-the-quality-of-code-comments","title":"Evaluating the Quality of Code Comments Generated by Large Language Models for Novice Programmers","date":"2024-09-22","arxiv_id":"2409.14368","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-model-agents-state-of-the-art","title":"Large Model Based Agents: State-of-the-Art, Cooperation Paradigms, Security and Privacy, and Future Trends","date":"2024-09-22","arxiv_id":"2409.14457","n_code_links":0,"syntology":null},{"paper":"/paper/2409-13989","slug":"2409-13989","title":"ChemEval: A Comprehensive Multi-Level Chemical Evaluation for Large Language Models","date":"2024-09-21","arxiv_id":"2409.13989","n_code_links":1,"syntology":{"ran":17,"of":18,"n_ran_checked":17,"n_instrument":0,"unverified":1,"pointer_only":18,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ustc-starteam/chemeval"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":17,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/2409-14037","slug":"2409-14037","title":"Can LLMs replace Neil deGrasse Tyson? Evaluating the Reliability of LLMs as Science Communicators","date":"2024-09-21","arxiv_id":"2409.14037","n_code_links":1,"syntology":null},{"paper":"/paper/aligning-language-models-using-follow-up","slug":"aligning-language-models-using-follow-up","title":"Aligning Language Models Using Follow-up Likelihood as Reward Signal","date":"2024-09-20","arxiv_id":"2409.13948","n_code_links":1,"syntology":null},{"paper":null,"slug":"prompting-large-language-models-for-1","title":"Prompting Large Language Models for Supporting the Differential Diagnosis of Anemia","date":"2024-09-20","arxiv_id":"2409.15377","n_code_links":0,"syntology":null},{"paper":"/paper/shizishangpt-an-agricultural-large-language","slug":"shizishangpt-an-agricultural-large-language","title":"ShizishanGPT: An Agricultural Large Language Model Integrating Tools and Resources","date":"2024-09-20","arxiv_id":"2409.13537","n_code_links":1,"syntology":null},{"paper":"/paper/stop-benchmarking-large-language-models-with","slug":"stop-benchmarking-large-language-models-with","title":"STOP! Benchmarking Large Language Models with Sensitivity Testing on Offensive Progressions","date":"2024-09-20","arxiv_id":"2409.13843","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-tinybert-for-financial-sentiment-1","slug":"enhancing-tinybert-for-financial-sentiment-1","title":"Enhancing TinyBERT for Financial Sentiment Analysis Using GPT-Augmented FinBERT Distillation","date":"2024-09-19","arxiv_id":"2409.18999","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-image-hallucination-in-text-to","slug":"evaluating-image-hallucination-in-text-to","title":"Evaluating Image Hallucination in Text-to-Image Generation with Question-Answering","date":"2024-09-19","arxiv_id":"2409.12784","n_code_links":1,"syntology":null},{"paper":null,"slug":"prompts-are-programs-too-understanding-how","title":"Prompts Are Programs Too! Understanding How Developers Build Software Containing Prompts","date":"2024-09-19","arxiv_id":"2409.12447","n_code_links":0,"syntology":null},{"paper":null,"slug":"taco-rl-task-aware-prompt-compression","title":"TACO-RL: Task Aware Prompt Compression Optimization with Reinforcement Learning","date":"2024-09-19","arxiv_id":"2409.13035","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-would-you-ask-when-you-first-saw-a-2-b-2","title":"What Would You Ask When You First Saw $a^2+b^2=c^2$? Evaluating LLM on Curiosity-Driven Questioning","date":"2024-09-19","arxiv_id":"2409.17172","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-lists-to-emojis-how-format-bias-affects","title":"From Lists to Emojis: How Format Bias Affects Model Alignment","date":"2024-09-18","arxiv_id":"2409.11704","n_code_links":0,"syntology":null},{"paper":null,"slug":"harnessing-llms-for-api-interactions-a","title":"Harnessing LLMs for API Interactions: A Framework for Classification and Synthetic Data Generation","date":"2024-09-18","arxiv_id":"2409.11703","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-context-faithfulness-in-large","title":"Investigating Context-Faithfulness in Large Language Models: The Roles of Memory Strength and Evidence Style","date":"2024-09-17","arxiv_id":"2409.10955","n_code_links":0,"syntology":null},{"paper":null,"slug":"sparks-of-artificial-general-intelligence-agi","title":"Sparks of Artificial General Intelligence(AGI) in Semiconductor Material Science: Early Explorations into the Next Frontier of Generative AI-Assisted Electron Micrograph Analysis","date":"2024-09-17","arxiv_id":"2409.12244","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-takes-the-sat-tracing-changes-in-test","title":"GPT takes the SAT: Tracing changes in Test Difficulty and Math Performance of Students","date":"2024-09-16","arxiv_id":"2409.10750","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-for-clinical-risk-prediction","title":"LLMs for clinical risk prediction","date":"2024-09-16","arxiv_id":"2409.10191","n_code_links":0,"syntology":null},{"paper":null,"slug":"mindguard-towards-accessible-and-sitgma-free","title":"MindGuard: Towards Accessible and Sitgma-free Mental Health First Aid via Edge LLM","date":"2024-09-16","arxiv_id":"2409.10064","n_code_links":0,"syntology":null},{"paper":"/paper/select-sql-self-correcting-ensemble-chain-of","slug":"select-sql-self-correcting-ensemble-chain-of","title":"SelECT-SQL: Self-correcting ensemble Chain-of-Thought for Text-to-SQL","date":"2024-09-16","arxiv_id":"2409.10007","n_code_links":1,"syntology":null}],"record_sha256":"7c45c1b0ceab800740dd15b25d9d1c10e00ae65ebe9944403204402da5d93fb1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}