{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/2","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":29,"rows_per_page":100,"rows":[101,200],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4","next":"/method/gpt-4/papers/3","papers":[{"paper":null,"slug":"utilizing-llms-to-investigate-the-disputed","title":"Utilizing LLMs to Investigate the Disputed Role of Evidence in Electronic Cigarette Health Policy Formation in Australia and the UK","date":"2025-05-10","arxiv_id":"2505.06782","n_code_links":0,"syntology":null},{"paper":null,"slug":"camera-control-at-the-edge-with-language","title":"Camera Control at the Edge with Language Models for Scene Understanding","date":"2025-05-09","arxiv_id":"2505.06402","n_code_links":0,"syntology":null},{"paper":null,"slug":"healthy-llms-benchmarking-llm-knowledge-of-uk","title":"Healthy LLMs? Benchmarking LLM Knowledge of UK Government Public Health Information","date":"2025-05-09","arxiv_id":"2505.06046","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-approaches-to-qualitative-and-quantitative","title":"AI Approaches to Qualitative and Quantitative News Analytics on NATO Unity","date":"2025-05-08","arxiv_id":"2505.06313","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-vision-language-action-models-in","slug":"benchmarking-vision-language-action-models-in","title":"Benchmarking Vision, Language, & Action Models in Procedurally Generated, Open Ended Action Environments","date":"2025-05-08","arxiv_id":"2505.05540","n_code_links":1,"syntology":null},{"paper":null,"slug":"performance-evaluation-of-large-language-1","title":"Performance Evaluation of Large Language Models in Bangla Consumer Health Query Summarization","date":"2025-05-08","arxiv_id":"2505.05070","n_code_links":0,"syntology":null},{"paper":null,"slug":"hiperrag-high-performance-retrieval-augmented","title":"HiPerRAG: High-Performance Retrieval Augmented Generation for Scientific Insights","date":"2025-05-07","arxiv_id":"2505.04846","n_code_links":0,"syntology":null},{"paper":null,"slug":"red-teaming-the-mind-of-the-machine-a","title":"Red Teaming the Mind of the Machine: A Systematic Evaluation of Prompt Injection and Jailbreak Vulnerabilities in LLMs","date":"2025-05-07","arxiv_id":"2505.04806","n_code_links":0,"syntology":null},{"paper":"/paper/llm-optira-llm-driven-optimization-of","slug":"llm-optira-llm-driven-optimization-of","title":"LLM-OptiRA: LLM-Driven Optimization of Resource Allocation for Non-Convex Problems in Wireless Communications","date":"2025-05-04","arxiv_id":"2505.02091","n_code_links":1,"syntology":null},{"paper":null,"slug":"seval-ex-a-statement-level-framework-for","title":"SEval-Ex: A Statement-Level Framework for Explainable Summarization Evaluation","date":"2025-05-04","arxiv_id":"2505.02235","n_code_links":0,"syntology":null},{"paper":null,"slug":"semantic-intelligence-integrating-gpt-4-with","title":"Semantic Intelligence: Integrating GPT-4 with A Planning in Low-Cost Robotics","date":"2025-05-03","arxiv_id":"2505.01931","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-sparql-query-rewriting-for-complex","title":"Enhancing SPARQL Query Rewriting for Complex Ontology Alignments","date":"2025-05-02","arxiv_id":"2505.01309","n_code_links":0,"syntology":null},{"paper":null,"slug":"good-news-for-script-kiddies-evaluating-large","title":"Good News for Script Kiddies? Evaluating Large Language Models for Automated Exploit Generation","date":"2025-05-02","arxiv_id":"2505.01065","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-document-level-biomedical-relation","title":"Zero-Shot Document-Level Biomedical Relation Extraction via Scenario-based Prompt Design in Two-Stage with LLM","date":"2025-05-02","arxiv_id":"2505.01077","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-security-and-strengthening-defenses","title":"Enhancing Security and Strengthening Defenses in Automated Short-Answer Grading Systems","date":"2025-04-30","arxiv_id":"2505.00061","n_code_links":0,"syntology":null},{"paper":"/paper/geolocating-earth-imagery-from-iss","slug":"geolocating-earth-imagery-from-iss","title":"Geolocating Earth Imagery from ISS: Integrating Machine Learning with Astronaut Photography for Enhanced Geographic Mapping","date":"2025-04-29","arxiv_id":"2504.21194","n_code_links":1,"syntology":null},{"paper":null,"slug":"jaccdiv-a-metric-and-benchmark-for","title":"JaccDiv: A Metric and Benchmark for Quantifying Diversity of Generated Marketing Text in the Music Industry","date":"2025-04-29","arxiv_id":"2504.20849","n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-large-language-models-for-medicine","title":"Multimodal Large Language Models for Medicine: A Comprehensive Survey","date":"2025-04-29","arxiv_id":"2504.21051","n_code_links":0,"syntology":null},{"paper":null,"slug":"yochameleon-personalized-vision-and-language","title":"YoChameleon: Personalized Vision and Language Generation","date":"2025-04-29","arxiv_id":"2504.20998","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-systematic-reviews-with-large","title":"Enhancing Systematic Reviews with Large Language Models: Using GPT-4 and Kimi","date":"2025-04-28","arxiv_id":"2504.20276","n_code_links":0,"syntology":null},{"paper":null,"slug":"m-kailin-knowledge-driven-agentic-scientific","title":"m-KAILIN: Knowledge-Driven Agentic Scientific Corpus Distillation Framework for Biomedical Large Language Models Training","date":"2025-04-28","arxiv_id":"2504.19565","n_code_links":0,"syntology":null},{"paper":"/paper/from-inductive-to-deductive-llms-based","slug":"from-inductive-to-deductive-llms-based","title":"From Inductive to Deductive: LLMs-Based Qualitative Data Analysis in Requirements Engineering","date":"2025-04-27","arxiv_id":"2504.19384","n_code_links":1,"syntology":null},{"paper":null,"slug":"why-you-shouldn-t-fully-trust-chatgpt-a","title":"Why you shouldn't fully trust ChatGPT: A synthesis of this AI tool's error rates across disciplines and the software engineering lifecycle","date":"2025-04-26","arxiv_id":"2504.18858","n_code_links":0,"syntology":null},{"paper":null,"slug":"advanced-chest-x-ray-analysis-via-transformer","title":"Advanced Chest X-Ray Analysis via Transformer-Based Image Descriptors and Cross-Model Attention Mechanism","date":"2025-04-23","arxiv_id":"2504.16774","n_code_links":0,"syntology":null},{"paper":null,"slug":"amplified-vulnerabilities-structured","title":"Amplified Vulnerabilities: Structured Jailbreak Attacks on LLM-based Multi-Agent Debate","date":"2025-04-23","arxiv_id":"2504.16489","n_code_links":0,"syntology":null},{"paper":null,"slug":"emo-pillars-knowledge-distillation-to-support","title":"Emo Pillars: Knowledge Distillation to Support Fine-Grained Context-Aware and Context-Less Emotion Classification","date":"2025-04-23","arxiv_id":"2504.16856","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-llms-as-meta-judges-a-multi-agent","title":"Leveraging LLMs as Meta-Judges: A Multi-Agent Framework for Evaluating LLM Judgments","date":"2025-04-23","arxiv_id":"2504.17087","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-large-scale-class-level-benchmark-dataset","title":"A Large-scale Class-level Benchmark Dataset for Code Generation with LLMs","date":"2025-04-22","arxiv_id":"2504.15564","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-as-data-annotators-how-close-are-we-to","title":"LLMs as Data Annotators: How Close Are We to Human Performance","date":"2025-04-21","arxiv_id":"2504.15022","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-framework-for-benchmarking-and-aligning","title":"A Framework for Benchmarking and Aligning Task-Planning Safety in LLM-Based Embodied Agents","date":"2025-04-20","arxiv_id":"2504.14650","n_code_links":0,"syntology":null},{"paper":"/paper/know-me-respond-to-me-benchmarking-llms-for","slug":"know-me-respond-to-me-benchmarking-llms-for","title":"Know Me, Respond to Me: Benchmarking LLMs for Dynamic User Profiling and Personalized Responses at Scale","date":"2025-04-19","arxiv_id":"2504.14225","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["bowen-upenn/personamem"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"simplifymytext-an-llm-based-system-for","title":"SimplifyMyText: An LLM-Based System for Inclusive Plain Language Text Simplification","date":"2025-04-19","arxiv_id":"2504.14223","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-sensitivity-evaluation-framework-for","title":"LLM Sensitivity Evaluation Framework for Clinical Diagnosis","date":"2025-04-18","arxiv_id":"2504.13475","n_code_links":0,"syntology":null},{"paper":null,"slug":"accuracy-is-not-agreement-expert-aligned","title":"Accuracy is Not Agreement: Expert-Aligned Evaluation of Crash Narrative Classification Models","date":"2025-04-17","arxiv_id":"2504.13068","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-expert-failures-improves-llm-agent","title":"Exploring Expert Failures Improves LLM Agent Tuning","date":"2025-04-17","arxiv_id":"2504.13145","n_code_links":0,"syntology":null},{"paper":null,"slug":"mitigating-llm-hallucinations-with-knowledge","title":"Mitigating LLM Hallucinations with Knowledge Graphs: A Case Study","date":"2025-04-16","arxiv_id":"2504.12422","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-role-of-kg-based-rag-in","title":"Exploring the Role of Knowledge Graph-Based RAG in Japanese Medical Question Answering with Small-Scale LLMs","date":"2025-04-15","arxiv_id":"2504.10982","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-chains-of-thought-benchmarking-latent","title":"Beyond Chains of Thought: Benchmarking Latent-Space Reasoning Abilities in Large Language Models","date":"2025-04-14","arxiv_id":"2504.10615","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-llms-handle-webshell-detection-overcoming","title":"Can LLMs handle WebShell detection? Overcoming Detection Challenges with Behavioral Function-Aware Framework","date":"2025-04-14","arxiv_id":"2504.13811","n_code_links":0,"syntology":null},{"paper":null,"slug":"emafusion-a-self-optimizing-system-for","title":"EMAFusion: A Self-Optimizing System for Seamless LLM Selection and Integration","date":"2025-04-14","arxiv_id":"2504.10681","n_code_links":0,"syntology":null},{"paper":null,"slug":"keyword-extraction-and-aspect-classification","title":"Keyword Extraction, and Aspect Classification in Sinhala, English, and Code-Mixed Content","date":"2025-04-14","arxiv_id":"2504.10679","n_code_links":0,"syntology":null},{"paper":"/paper/clinicalgpt-r1-pushing-reasoning-capability","slug":"clinicalgpt-r1-pushing-reasoning-capability","title":"ClinicalGPT-R1: Pushing reasoning capability of generalist disease diagnosis with large language model","date":"2025-04-13","arxiv_id":"2504.09421","n_code_links":1,"syntology":null},{"paper":null,"slug":"integrating-large-language-models-for-1","title":"Integrating Large Language Models for Automated Structural Analysis","date":"2025-04-13","arxiv_id":"2504.09754","n_code_links":0,"syntology":null},{"paper":null,"slug":"iterative-self-training-for-code-generation","title":"Iterative Self-Training for Code Generation via Reinforced Re-Ranking","date":"2025-04-13","arxiv_id":"2504.09643","n_code_links":0,"syntology":null},{"paper":null,"slug":"llmtaxo-leveraging-large-language-models-for","title":"LLMTaxo: Leveraging Large Language Models for Constructing Taxonomy of Factual Claims from Social Media","date":"2025-04-11","arxiv_id":"2504.12325","n_code_links":0,"syntology":null},{"paper":null,"slug":"rtlrepocoder-repository-level-rtl-code","title":"RTLRepoCoder: Repository-Level RTL Code Completion through the Combination of Fine-Tuning and Retrieval Augmentation","date":"2025-04-11","arxiv_id":"2504.08862","n_code_links":0,"syntology":null},{"paper":null,"slug":"attentiondefense-leveraging-system-prompt","title":"AttentionDefense: Leveraging System Prompt Attention for Explainable Defense Against Novel Jailbreaks","date":"2025-04-10","arxiv_id":"2504.12321","n_code_links":0,"syntology":null},{"paper":null,"slug":"has-the-creativity-of-large-language-models","title":"Has the Creativity of Large-Language Models peaked? An analysis of inter- and intra-LLM variability","date":"2025-04-10","arxiv_id":"2504.12320","n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-prompt-optimization-with-large","title":"Revisiting Prompt Optimization with Large Reasoning Models-A Case Study on Event Extraction","date":"2025-04-10","arxiv_id":"2504.07357","n_code_links":0,"syntology":null},{"paper":null,"slug":"synthetic-fluency-hallucinations","title":"Synthetic Fluency: Hallucinations, Confabulations, and the Creation of Irish Words in LLM-Generated Translations","date":"2025-04-10","arxiv_id":"2504.07680","n_code_links":0,"syntology":null},{"paper":null,"slug":"endowing-embodied-agents-with-spatial","title":"Endowing Embodied Agents with Spatial Reasoning Capabilities for Vision-and-Language Navigation","date":"2025-04-09","arxiv_id":"2504.08806","n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-how-text-to-image-models-represent","title":"Text-to-Image Models and Their Representation of People from Different Nationalities Engaging in Activities","date":"2025-04-08","arxiv_id":"2504.06313","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-how-hyperparameters-impact-large","title":"Assessing how hyperparameters impact Large Language Models' sarcasm detection performance","date":"2025-04-08","arxiv_id":"2504.06166","n_code_links":0,"syntology":null},{"paper":null,"slug":"instructionbench-an-instructional-video","title":"InstructionBench: An Instructional Video Understanding Benchmark","date":"2025-04-07","arxiv_id":"2504.05040","n_code_links":0,"syntology":null},{"paper":null,"slug":"provable-failure-of-language-models-in","title":"Provable Failure of Language Models in Learning Majority Boolean Logic via Gradient Descent","date":"2025-04-07","arxiv_id":"2504.04702","n_code_links":0,"syntology":null},{"paper":null,"slug":"cola-learning-to-interactively-collaborate","title":"CoLa -- Learning to Interactively Collaborate with Large LMs","date":"2025-04-03","arxiv_id":"2504.02965","n_code_links":0,"syntology":null},{"paper":null,"slug":"learnat-learning-nl2sql-with-ast-guided-task","title":"LearNAT: Learning NL2SQL with AST-guided Task Decomposition for Large Language Models","date":"2025-04-03","arxiv_id":"2504.02327","n_code_links":0,"syntology":null},{"paper":"/paper/task-as-context-prompting-for-accurate","slug":"task-as-context-prompting-for-accurate","title":"Task as Context Prompting for Accurate Medical Symptom Coding Using Large Language Models","date":"2025-04-03","arxiv_id":"2504.03051","n_code_links":1,"syntology":null},{"paper":null,"slug":"pico-jailbreaking-multimodal-large-language","title":"PiCo: Jailbreaking Multimodal Large Language Models via $\\textbf{Pi}$ctorial $\\textbf{Co}$de Contextualization","date":"2025-04-02","arxiv_id":"2504.01444","n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-factual-benchmarking-for-in-car","title":"Automated Factual Benchmarking for In-Car Conversational Systems using Large Language Models","date":"2025-04-01","arxiv_id":"2504.01248","n_code_links":0,"syntology":null},{"paper":null,"slug":"collaborative-llm-numerical-reasoning-with","title":"Collaborative LLM Numerical Reasoning with Local Data Protection","date":"2025-04-01","arxiv_id":"2504.00299","n_code_links":0,"syntology":null},{"paper":null,"slug":"srlcg-self-rectified-large-scale-code","title":"SRLCG: Self-Rectified Large-Scale Code Generation with Multidimensional Chain-of-Thought and Dynamic Backtracking","date":"2025-04-01","arxiv_id":"2504.00532","n_code_links":0,"syntology":null},{"paper":null,"slug":"judgelrm-large-reasoning-models-as-a-judge","title":"JudgeLRM: Large Reasoning Models as a Judge","date":"2025-03-31","arxiv_id":"2504.00050","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-pass-the-turing-test","title":"Large Language Models Pass the Turing Test","date":"2025-03-31","arxiv_id":"2503.23674","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm4fs-leveraging-large-language-models-for","title":"LLM4FS: Leveraging Large Language Models for Feature Selection and How to Improve It","date":"2025-03-31","arxiv_id":"2503.24157","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-gpt-4-for-robotic-agent-strategy","title":"Exploring GPT-4 for Robotic Agent Strategy with Real-Time State Feedback and a Reactive Behaviour Framework","date":"2025-03-30","arxiv_id":"2503.23601","n_code_links":0,"syntology":null},{"paper":null,"slug":"ferg-llm-feature-engineering-by-reason","title":"FeRG-LLM : Feature Engineering by Reason Generation Large Language Models","date":"2025-03-30","arxiv_id":"2503.23371","n_code_links":0,"syntology":null},{"paper":"/paper/rare-retrieval-augmented-reasoning-modeling","slug":"rare-retrieval-augmented-reasoning-modeling","title":"RARE: Retrieval-Augmented Reasoning Modeling","date":"2025-03-30","arxiv_id":"2503.23513","n_code_links":1,"syntology":{"ran":13,"of":14,"n_ran_checked":13,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["open-dataflow/rare"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-training-free-llm-framework-with","title":"A Training-free LLM Framework with Interaction between Contextually Related Subtasks in Solving Complex Tasks","date":"2025-03-29","arxiv_id":"2503.23053","n_code_links":0,"syntology":null},{"paper":"/paper/multimodal-machine-learning-with-large","slug":"multimodal-machine-learning-with-large","title":"Multimodal machine learning with large language embedding model for polymer property prediction","date":"2025-03-29","arxiv_id":"2503.22962","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-well-can-vison-language-models-understand","title":"How Well Can Vison-Language Models Understand Humans' Intention? An Open-ended Theory of Mind Question Evaluation Benchmark","date":"2025-03-28","arxiv_id":"2503.22093","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrating-artificial-intelligence-with","title":"Integrating Artificial Intelligence with Human Expertise: An In-depth Analysis of ChatGPT's Capabilities in Generating Metamorphic Relations","date":"2025-03-28","arxiv_id":"2503.22141","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-inequality-of-llm-fact-checking","title":"Understanding Inequality of LLM Fact-Checking over Geographic Regions with Agent and Retrieval models","date":"2025-03-28","arxiv_id":"2503.22877","n_code_links":0,"syntology":null},{"paper":null,"slug":"collab-controlled-decoding-using-mixture-of","title":"Collab: Controlled Decoding using Mixture of Agents for LLM Alignment","date":"2025-03-27","arxiv_id":"2503.21720","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-large-language-models-to-produce","title":"Using large language models to produce literature reviews: Usages and systematic biases of microphysics parametrizations in 2699 publications","date":"2025-03-27","arxiv_id":"2503.21352","n_code_links":0,"syntology":null},{"paper":"/paper/vision-language-models-versus-machine","slug":"vision-language-models-versus-machine","title":"Vision Language Models versus Machine Learning Models Performance on Polyp Detection and Classification in Colonoscopy Images","date":"2025-03-27","arxiv_id":"2503.21840","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-we-make-code-green-understanding-trade","title":"Can We Make Code Green? Understanding Trade-Offs in LLMs vs. Human Code Optimizations","date":"2025-03-26","arxiv_id":"2503.20126","n_code_links":0,"syntology":null},{"paper":null,"slug":"iterative-prompting-with-persuasion-skills-in","title":"Iterative Prompting with Persuasion Skills in Jailbreaking Large Language Models","date":"2025-03-26","arxiv_id":"2503.20320","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-aware-semantic-segmentation-enhancing","title":"Context-Aware Semantic Segmentation: Enhancing Pixel-Level Understanding with Large Language Models for Advanced Vision Applications","date":"2025-03-25","arxiv_id":"2503.19276","n_code_links":0,"syntology":null},{"paper":"/paper/enabling-rapid-shared-human-ai-mental-model","slug":"enabling-rapid-shared-human-ai-mental-model","title":"Enabling Rapid Shared Human-AI Mental Model Alignment via the After-Action Review","date":"2025-03-25","arxiv_id":"2503.19607","n_code_links":1,"syntology":null},{"paper":"/paper/fundamental-limits-of-perfect-concept-erasure","slug":"fundamental-limits-of-perfect-concept-erasure","title":"Fundamental Limits of Perfect Concept Erasure","date":"2025-03-25","arxiv_id":"2503.20098","n_code_links":1,"syntology":null},{"paper":null,"slug":"sci-idea-context-aware-scientific-ideation","title":"SCI-IDEA: Context-Aware Scientific Ideation Using Token and Sentence Embeddings","date":"2025-03-25","arxiv_id":"2503.19257","n_code_links":0,"syntology":null},{"paper":null,"slug":"taxonomy-inference-for-tabular-data-using","title":"Taxonomy Inference for Tabular Data Using Large Language Models","date":"2025-03-25","arxiv_id":"2503.21810","n_code_links":0,"syntology":null},{"paper":"/paper/global-local-tree-search-for-language-guided","slug":"global-local-tree-search-for-language-guided","title":"Global-Local Tree Search in VLMs for 3D Indoor Scene Generation","date":"2025-03-24","arxiv_id":"2503.18476","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dw-dengwei/treesearchgen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"tdri-two-phase-dialogue-refinement-and-co","title":"TDRI: Two-Phase Dialogue Refinement and Co-Adaptation for Interactive Image Generation","date":"2025-03-22","arxiv_id":"2503.17669","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-the-reliability-and-validity-of-gpt","title":"Assessing the Reliability and Validity of GPT-4 in Annotating Emotion Appraisal Ratings","date":"2025-03-21","arxiv_id":"2503.16883","n_code_links":0,"syntology":null},{"paper":null,"slug":"coke-customizable-fine-grained-story","title":"CoKe: Customizable Fine-Grained Story Evaluation via Chain-of-Keyword Rationalization","date":"2025-03-21","arxiv_id":"2503.17136","n_code_links":0,"syntology":null},{"paper":null,"slug":"saudiculture-a-benchmark-for-evaluating-large","title":"SaudiCulture: A Benchmark for Evaluating Large Language Models Cultural Competence within Saudi Arabia","date":"2025-03-21","arxiv_id":"2503.17485","n_code_links":0,"syntology":null},{"paper":"/paper/when-words-outperform-vision-vlms-can-self","slug":"when-words-outperform-vision-vlms-can-self","title":"When Words Outperform Vision: VLMs Can Self-Improve Via Text-Only Training For Human-Centered Decision Making","date":"2025-03-21","arxiv_id":"2503.16965","n_code_links":0,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"the-lighthouse-of-language-enhancing-llm","title":"The Lighthouse of Language: Enhancing LLM Agents via Critique-Guided Improvement","date":"2025-03-20","arxiv_id":"2503.16024","n_code_links":0,"syntology":null},{"paper":"/paper/eltex-a-framework-for-domain-driven-synthetic","slug":"eltex-a-framework-for-domain-driven-synthetic","title":"ELTEX: A Framework for Domain-Driven Synthetic Data Generation","date":"2025-03-19","arxiv_id":"2503.15055","n_code_links":1,"syntology":null},{"paper":null,"slug":"truthlens-a-training-free-paradigm-for","title":"TruthLens:A Training-Free Paradigm for DeepFake Detection","date":"2025-03-19","arxiv_id":"2503.15342","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-the-generalization-of-in","slug":"understanding-the-generalization-of-in","title":"Understanding the Generalization of In-Context Learning in Transformers: An Empirical Study","date":"2025-03-19","arxiv_id":"2503.15579","n_code_links":1,"syntology":null},{"paper":"/paper/gricean-norms-as-a-basis-for-effective","slug":"gricean-norms-as-a-basis-for-effective","title":"Gricean Norms as a Basis for Effective Collaboration","date":"2025-03-18","arxiv_id":"2503.14484","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-for-virtual-human","title":"Large Language Models for Virtual Human Gesture Selection","date":"2025-03-18","arxiv_id":"2503.14408","n_code_links":0,"syntology":null},{"paper":"/paper/pencil-long-thoughts-with-short-memory","slug":"pencil-long-thoughts-with-short-memory","title":"PENCIL: Long Thoughts with Short Memory","date":"2025-03-18","arxiv_id":"2503.14337","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chr26195/pencil"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fragile-mastery-are-domain-specific-trade","title":"Fragile Mastery: Are Domain-Specific Trade-Offs Undermining On-Device Language Models?","date":"2025-03-16","arxiv_id":"2503.22698","n_code_links":0,"syntology":null},{"paper":"/paper/llm-hpc-benchmarking-deepseek-s-performance","slug":"llm-hpc-benchmarking-deepseek-s-performance","title":"LLM & HPC:Benchmarking DeepSeek's Performance in High-Performance Computing Tasks","date":"2025-03-15","arxiv_id":"2504.03665","n_code_links":1,"syntology":null},{"paper":null,"slug":"maritime-mission-planning-for-unmanned","title":"Maritime Mission Planning for Unmanned Surface Vessel using Large Language Model","date":"2025-03-15","arxiv_id":"2503.12065","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-sentiment-the-catalyst-for-llm-change","title":"Prompt Sentiment: The Catalyst for LLM Change","date":"2025-03-14","arxiv_id":"2503.13510","n_code_links":0,"syntology":null}],"record_sha256":"b044746e7558a477413eeb25a37bc0050670d9134f54115ccf8bef16a4f5187b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}