{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt/papers/3","list_of":"/method/gpt","method":"GPT","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":3,"pages_in_order":13,"rows_per_page":100,"rows":[201,300],"of":1212,"counts":{"archive_papers_tagged":1212,"with_a_code_link":453,"where_syntology_ran_a_sample":152,"not_listed_spam_title":0,"listed":1212,"listed_where_code_ran":152,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":130,"every_run_a_failure_of_syntologys_instrument":22,"listed_with_a_run_with_no_instrument_failure":130,"listed_every_run_a_failure_of_syntologys_instrument":22,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt","prev":"/method/gpt/papers/2","next":"/method/gpt/papers/4","papers":[{"paper":null,"slug":"the-rosetta-paradox-domain-specific","title":"The Rosetta Paradox: Domain-Specific Performance Inversions in Large Language Models","date":"2024-12-09","arxiv_id":"2412.17821","n_code_links":0,"syntology":null},{"paper":"/paper/characterbox-evaluating-the-role-playing","slug":"characterbox-evaluating-the-role-playing","title":"CharacterBox: Evaluating the Role-Playing Capabilities of LLMs in Text-Based Virtual Worlds","date":"2024-12-07","arxiv_id":"2412.05631","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["paitesanshi/characterbox"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-the-use-of-llms-for-sql-equivalence","title":"Can the Rookies Cut the Tough Cookie? Exploring the Use of LLMs for SQL Equivalence Checking","date":"2024-12-07","arxiv_id":"2412.05561","n_code_links":0,"syntology":null},{"paper":"/paper/privagent-agentic-based-red-teaming-for-llm","slug":"privagent-agentic-based-red-teaming-for-llm","title":"PrivAgent: Agentic-based Red-teaming for LLM Privacy Leakage","date":"2024-12-07","arxiv_id":"2412.05734","n_code_links":1,"syntology":null},{"paper":null,"slug":"are-frontier-large-language-models-suitable","title":"Are Frontier Large Language Models Suitable for Q&A in Science Centres?","date":"2024-12-06","arxiv_id":"2412.05200","n_code_links":0,"syntology":null},{"paper":null,"slug":"queen-a-large-language-model-for-quechua","title":"QueEn: A Large Language Model for Quechua-English Translation","date":"2024-12-06","arxiv_id":"2412.05184","n_code_links":0,"syntology":null},{"paper":null,"slug":"compressing-kv-cache-for-long-context-llm","title":"Compressing KV Cache for Long-Context LLM Inference with Inter-Layer Attention Similarity","date":"2024-12-03","arxiv_id":"2412.02252","n_code_links":0,"syntology":null},{"paper":null,"slug":"flattering-to-deceive-the-impact-of","title":"Flattering to Deceive: The Impact of Sycophantic Behavior on User Trust in Large Language Model","date":"2024-12-03","arxiv_id":"2412.02802","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-promise-and-peril-of-generative-ai","title":"The Promise and Peril of Generative AI: Evidence from GPT-4 as Sell-Side Analysts","date":"2024-12-02","arxiv_id":"2412.01069","n_code_links":0,"syntology":null},{"paper":"/paper/a-comprehensive-guide-to-explainable-ai-from","slug":"a-comprehensive-guide-to-explainable-ai-from","title":"A Comprehensive Guide to Explainable AI: From Classical Models to LLMs","date":"2024-12-01","arxiv_id":"2412.00800","n_code_links":1,"syntology":null},{"paper":null,"slug":"eventgpt-event-stream-understanding-with","title":"EventGPT: Event Stream Understanding with Multimodal Large Language Models","date":"2024-12-01","arxiv_id":"2412.00832","n_code_links":0,"syntology":null},{"paper":null,"slug":"cdemapper-enhancing-nih-common-data-element","title":"CDEMapper: Enhancing NIH Common Data Element Normalization using Large Language Models","date":"2024-11-30","arxiv_id":"2412.00491","n_code_links":0,"syntology":null},{"paper":null,"slug":"beautimeter-harnessing-gpt-for-assessing","title":"Beautimeter: Harnessing GPT for Assessing Architectural and Urban Beauty based on the 15 Properties of Living Structure","date":"2024-11-28","arxiv_id":"2411.19094","n_code_links":0,"syntology":null},{"paper":null,"slug":"habit-coach-customising-rag-based-chatbots-to","title":"Habit Coach: Customising RAG-based chatbots to support behavior change","date":"2024-11-28","arxiv_id":"2411.19229","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impact-of-example-selection-in-few-shot","title":"The Impact of Example Selection in Few-Shot Prompting on Automated Essay Scoring Using GPT Models","date":"2024-11-28","arxiv_id":"2411.18924","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-bidirectional-encoder-become-the-ultimate","title":"Can bidirectional encoder become the ultimate winner for downstream applications of foundation models?","date":"2024-11-27","arxiv_id":"2411.18021","n_code_links":0,"syntology":null},{"paper":null,"slug":"chatgpt-as-speechwriter-for-the-french","title":"ChatGPT as speechwriter for the French presidents","date":"2024-11-27","arxiv_id":"2411.18382","n_code_links":0,"syntology":null},{"paper":null,"slug":"advancing-content-moderation-evaluating-large","title":"Advancing Content Moderation: Evaluating Large Language Models for Detecting Sensitive Content Across Text, Images, and Videos","date":"2024-11-26","arxiv_id":"2411.17123","n_code_links":0,"syntology":null},{"paper":null,"slug":"give-me-the-code-log-analysis-of-first-year","title":"\"Give me the code\" -- Log Analysis of First-Year CS Students' Interactions With GPT","date":"2024-11-26","arxiv_id":"2411.17855","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-limitations-of-llm-as-annotator-for-low","title":"On Limitations of LLM as Annotator for Low Resource Languages","date":"2024-11-26","arxiv_id":"2411.17637","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-transformers-truly-foundational-for","title":"Are Transformers Truly Foundational for Robotics?","date":"2024-11-25","arxiv_id":"2411.16917","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-ai-grade-your-essays-a-comparative","title":"Can AI grade your essays? A comparative analysis of large language models and teacher ratings in multidimensional essay scoring","date":"2024-11-25","arxiv_id":"2411.16337","n_code_links":0,"syntology":null},{"paper":"/paper/marketgpt-developing-a-pre-trained","slug":"marketgpt-developing-a-pre-trained","title":"MarketGPT: Developing a Pre-trained transformer (GPT) for Modeling Financial Time Series","date":"2024-11-25","arxiv_id":"2411.16585","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aaron-wheeler/marketgpt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"predictive-power-of-llms-in-financial-markets","title":"Predictive Power of LLMs in Financial Markets","date":"2024-11-25","arxiv_id":"2411.16569","n_code_links":0,"syntology":null},{"paper":"/paper/all-that-glitters-approaches-to-evaluations","slug":"all-that-glitters-approaches-to-evaluations","title":"\"All that Glitters\": Approaches to Evaluations with Unreliable Model and Human Annotations","date":"2024-11-23","arxiv_id":"2411.15634","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-next-tokens-via-second-last","title":"Improving Next Tokens via Second-Last Predictions with Generate and Refine","date":"2024-11-23","arxiv_id":"2411.15661","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparative-analysis-of-pooling-mechanisms-in","title":"Comparative Analysis of Pooling Mechanisms in LLMs: A Sentiment Analysis Perspective","date":"2024-11-22","arxiv_id":"2411.14654","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessment-of-llm-responses-to-end-user","title":"Assessment of LLM Responses to End-user Security Questions","date":"2024-11-21","arxiv_id":"2411.14571","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-the-robustness-of-analogical","slug":"evaluating-the-robustness-of-analogical","title":"Evaluating the Robustness of Analogical Reasoning in Large Language Models","date":"2024-11-21","arxiv_id":"2411.14215","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["marthaflinderslewis/robust-analogy"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ai-driven-agents-with-prompts-designed-for","title":"AI-Driven Agents with Prompts Designed for High Agreeableness Increase the Likelihood of Being Mistaken for a Human in the Turing Test","date":"2024-11-20","arxiv_id":"2411.13749","n_code_links":0,"syntology":null},{"paper":"/paper/combining-autoregressive-and-autoencoder","slug":"combining-autoregressive-and-autoencoder","title":"Combining Autoregressive and Autoencoder Language Models for Text Classification","date":"2024-11-20","arxiv_id":"2411.13282","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-large-language-models-for-climate","title":"Exploring Large Language Models for Climate Forecasting","date":"2024-11-20","arxiv_id":"2411.13724","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-virtual-reality-and-ai-tutoring","title":"Leveraging Virtual Reality and AI Tutoring for Language Learning: A Case Study of a Virtual Campus Environment with OpenAI GPT Integration with Unity 3D","date":"2024-11-19","arxiv_id":"2411.12619","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-open-source-llms-enhance-data","title":"Can Open-source LLMs Enhance Data Synthesis for Toxic Detection?: An Experimental Study","date":"2024-11-18","arxiv_id":"2411.15175","n_code_links":0,"syntology":null},{"paper":"/paper/cnmbert-a-model-for-hanyu-pinyin-abbreviation","slug":"cnmbert-a-model-for-hanyu-pinyin-abbreviation","title":"CNMBERT: A Model for Converting Hanyu Pinyin Abbreviations to Chinese Characters","date":"2024-11-18","arxiv_id":"2411.11770","n_code_links":1,"syntology":null},{"paper":"/paper/versatune-fine-tuning-multi-ability-llms","slug":"versatune-fine-tuning-multi-ability-llms","title":"VersaTune: An Efficient Data Composition Framework for Training Multi-Capability LLMs","date":"2024-11-18","arxiv_id":"2411.11266","n_code_links":1,"syntology":null},{"paper":null,"slug":"does-prompt-formatting-have-any-impact-on-llm","title":"Does Prompt Formatting Have Any Impact on LLM Performance?","date":"2024-11-15","arxiv_id":"2411.10541","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-static-tools-evaluating-large-language","title":"Beyond Static Tools: Evaluating Large Language Models for Cryptographic Misuse Detection","date":"2024-11-14","arxiv_id":"2411.09772","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-app-squatting-and-cloning","title":"LLM App Squatting and Cloning","date":"2024-11-12","arxiv_id":"2411.07518","n_code_links":0,"syntology":null},{"paper":null,"slug":"explore-the-reasoning-capability-of-llms-in","title":"Explore the Reasoning Capability of LLMs in the Chess Testbed","date":"2024-11-11","arxiv_id":"2411.06655","n_code_links":0,"syntology":null},{"paper":"/paper/detecting-reference-errors-in-scientific","slug":"detecting-reference-errors-in-scientific","title":"Detecting Reference Errors in Scientific Literature with Large Language Models","date":"2024-11-09","arxiv_id":"2411.06101","n_code_links":1,"syntology":null},{"paper":null,"slug":"sufficient-context-a-new-lens-on-retrieval","title":"Sufficient Context: A New Lens on Retrieval Augmented Generation Systems","date":"2024-11-09","arxiv_id":"2411.06037","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-visual-classification-using","slug":"enhancing-visual-classification-using","title":"Enhancing Visual Classification using Comparative Descriptors","date":"2024-11-08","arxiv_id":"2411.05357","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-semantic-cache-reducing-llm-costs-and","title":"GPT Semantic Cache: Reducing LLM Costs and Latency via Semantic Embedding Caching","date":"2024-11-08","arxiv_id":"2411.05276","n_code_links":0,"syntology":null},{"paper":"/paper/learning-the-rules-of-peptide-self-assembly","slug":"learning-the-rules-of-peptide-self-assembly","title":"Learning the rules of peptide self-assembly through data mining with large language models","date":"2024-11-08","arxiv_id":"2411.05421","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-guided-monte-carlo-tree-search-for","title":"GPT-Guided Monte Carlo Tree Search for Symbolic Regression in Financial Fraud Detection","date":"2024-11-07","arxiv_id":"2411.04459","n_code_links":0,"syntology":null},{"paper":null,"slug":"selecting-between-bert-and-gpt-for-text","title":"Selecting Between BERT and GPT for Text Classification in Political Science Research","date":"2024-11-07","arxiv_id":"2411.05050","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-word-vectors-to-multimodal-embeddings","title":"From Word Vectors to Multimodal Embeddings: Techniques, Applications, and Future Directions For Large Language Models","date":"2024-11-06","arxiv_id":"2411.05036","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-device-emoji-classifier-trained-with-gpt","title":"On-Device Emoji Classifier Trained with GPT-based Data Augmentation for a Mobile Keyboard","date":"2024-11-06","arxiv_id":"2411.05031","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-the-effects-of-human-written","slug":"understanding-the-effects-of-human-written","title":"Understanding the Effects of Human-written Paraphrases in LLM-generated Text Detection","date":"2024-11-06","arxiv_id":"2411.03806","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-transformer-training-efficiency","title":"Enhancing Transformer Training Efficiency with Dynamic Dropout","date":"2024-11-05","arxiv_id":"2411.03236","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-benefits-of-domain-pretraining","title":"Exploring the Benefits of Domain-Pretraining of Generative Large Language Models for Chemistry","date":"2024-11-05","arxiv_id":"2411.03542","n_code_links":0,"syntology":null},{"paper":"/paper/ask-and-it-shall-be-given-turing-completeness","slug":"ask-and-it-shall-be-given-turing-completeness","title":"Ask, and it shall be given: On the Turing completeness of prompting","date":"2024-11-04","arxiv_id":"2411.01992","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-the-ability-of-large-language-1","title":"Evaluating the Ability of Large Language Models to Generate Verifiable Specifications in VeriFast","date":"2024-11-04","arxiv_id":"2411.02318","n_code_links":0,"syntology":null},{"paper":null,"slug":"mdeval-massively-multilingual-code-debugging","title":"MdEval: Massively Multilingual Code Debugging","date":"2024-11-04","arxiv_id":"2411.02310","n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-reducing-the-need-for-learning-rate","title":"Analyzing & Reducing the Need for Learning Rate Warmup in GPT Training","date":"2024-10-31","arxiv_id":"2410.23922","n_code_links":0,"syntology":null},{"paper":null,"slug":"automating-quantum-software-maintenance","title":"Automating Quantum Software Maintenance: Flakiness Detection and Root Cause Analysis","date":"2024-10-31","arxiv_id":"2410.23578","n_code_links":0,"syntology":null},{"paper":"/paper/semantic-enrichment-of-the-quantum-cascade","slug":"semantic-enrichment-of-the-quantum-cascade","title":"Semantic Enrichment of the Quantum Cascade Laser Properties in Text- A Knowledge Graph Generation Approach","date":"2024-10-30","arxiv_id":"2410.22996","n_code_links":1,"syntology":null},{"paper":null,"slug":"cfsafety-comprehensive-fine-grained-safety","title":"CFSafety: Comprehensive Fine-grained Safety Assessment for LLMs","date":"2024-10-29","arxiv_id":"2410.21695","n_code_links":0,"syntology":null},{"paper":null,"slug":"coupling-quantum-like-cognition-with-the","title":"Coupling quantum-like cognition with the neuronal networks within generalized probability theory","date":"2024-10-29","arxiv_id":"2411.00036","n_code_links":0,"syntology":null},{"paper":null,"slug":"factbench-a-dynamic-benchmark-for-in-the-wild","title":"FactBench: A Dynamic Benchmark for In-the-Wild Language Model Factuality Evaluation","date":"2024-10-29","arxiv_id":"2410.22257","n_code_links":0,"syntology":null},{"paper":null,"slug":"semantic-search-evaluation","title":"Semantic Search Evaluation","date":"2024-10-28","arxiv_id":"2410.21549","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-tutorial-on-teaching-data-analytics-with","title":"A Tutorial on Teaching Data Analytics with Generative AI","date":"2024-10-25","arxiv_id":"2411.07244","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrating-large-language-models-with-2","title":"Integrating Large Language Models with Internet of Things Applications","date":"2024-10-25","arxiv_id":"2410.19223","n_code_links":0,"syntology":null},{"paper":"/paper/little-giants-synthesizing-high-quality","slug":"little-giants-synthesizing-high-quality","title":"Little Giants: Synthesizing High-Quality Embedding Data at Scale","date":"2024-10-24","arxiv_id":"2410.18634","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["haon-chen/SPEED"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"probing-ranking-llms-mechanistic","title":"Understanding Ranking LLMs: A Mechanistic Analysis for Information Retrieval","date":"2024-10-24","arxiv_id":"2410.18527","n_code_links":0,"syntology":null},{"paper":null,"slug":"future-token-prediction-causal-language","title":"Future Token Prediction -- Causal Language Modelling with Per-Token Semantic State Vector for Multi-Token Prediction","date":"2024-10-23","arxiv_id":"2410.18160","n_code_links":0,"syntology":null},{"paper":"/paper/omniflatten-an-end-to-end-gpt-model-for","slug":"omniflatten-an-end-to-end-gpt-model-for","title":"OmniFlatten: An End-to-end GPT Model for Seamless Voice Conversation","date":"2024-10-23","arxiv_id":"2410.17799","n_code_links":1,"syntology":null},{"paper":"/paper/developing-retrieval-augmented-generation-rag","slug":"developing-retrieval-augmented-generation-rag","title":"Developing Retrieval Augmented Generation (RAG) based LLM Systems from PDFs: An Experience Report","date":"2024-10-21","arxiv_id":"2410.15944","n_code_links":1,"syntology":null},{"paper":null,"slug":"using-gpt-models-for-qualitative-and","title":"Using GPT Models for Qualitative and Quantitative News Analytics in the 2024 US Presidental Election Process","date":"2024-10-21","arxiv_id":"2410.15884","n_code_links":0,"syntology":null},{"paper":"/paper/does-chatgpt-have-a-poetic-style","slug":"does-chatgpt-have-a-poetic-style","title":"Does ChatGPT Have a Poetic Style?","date":"2024-10-20","arxiv_id":"2410.15299","n_code_links":1,"syntology":null},{"paper":null,"slug":"sdp4bit-toward-4-bit-communication","title":"SDP4Bit: Toward 4-bit Communication Quantization in Sharded Data Parallelism for LLM Training","date":"2024-10-20","arxiv_id":"2410.15526","n_code_links":0,"syntology":null},{"paper":null,"slug":"metacognitive-monitoring-a-human-ability","title":"Judgment of Learning: A Human Ability Beyond Generative Artificial Intelligence","date":"2024-10-17","arxiv_id":"2410.13392","n_code_links":0,"syntology":null},{"paper":null,"slug":"shapefilegpt-a-multi-agent-large-language","title":"ShapefileGPT: A Multi-Agent Large Language Model Framework for Automated Shapefile Processing","date":"2024-10-16","arxiv_id":"2410.12376","n_code_links":0,"syntology":null},{"paper":"/paper/stabilize-the-latent-space-for-image","slug":"stabilize-the-latent-space-for-image","title":"Stabilize the Latent Space for Image Autoregressive Modeling: A Unified Perspective","date":"2024-10-16","arxiv_id":"2410.12490","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["DAMO-NLP-SG/DiGIT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"when-not-to-answer-evaluating-prompts-on-gpt","title":"When Not to Answer: Evaluating Prompts on GPT Models for Effective Abstention in Unanswerable Math Word Problems","date":"2024-10-16","arxiv_id":"2410.13029","n_code_links":0,"syntology":null},{"paper":"/paper/deciphering-the-chaos-enhancing-jailbreak","slug":"deciphering-the-chaos-enhancing-jailbreak","title":"Deciphering the Chaos: Enhancing Jailbreak Attacks via Adversarial Prompt Translation","date":"2024-10-15","arxiv_id":"2410.11317","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qizhangli/adversarial-prompt-translator"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evidence-of-cognitive-deficits","title":"Evidence of Cognitive Deficits andDevelopmental Advances in Generative AI: A Clock Drawing Test Analysis","date":"2024-10-15","arxiv_id":"2410.11756","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-hate-lost-in-translation-evaluation-of","title":"\"Is Hate Lost in Translation?\": Evaluation of Multilingual LGBTQIA+ Hate Speech Detection","date":"2024-10-15","arxiv_id":"2410.11230","n_code_links":0,"syntology":null},{"paper":"/paper/mtu-bench-a-multi-granularity-tool-use","slug":"mtu-bench-a-multi-granularity-tool-use","title":"MTU-Bench: A Multi-granularity Tool-Use Benchmark for Large Language Models","date":"2024-10-15","arxiv_id":"2410.11710","n_code_links":1,"syntology":null},{"paper":null,"slug":"nonlinear-gaussian-process-tomography-with","title":"Nonlinear Gaussian process tomography with imposed non-negativity constraints on physical quantities for plasma diagnostics","date":"2024-10-15","arxiv_id":"2410.11454","n_code_links":0,"syntology":null},{"paper":"/paper/double-jeopardy-and-climate-impact-in-the-use","slug":"double-jeopardy-and-climate-impact-in-the-use","title":"Double Jeopardy and Climate Impact in the Use of Large Language Models: Socio-economic Disparities and Reduced Utility for Non-English Speakers","date":"2024-10-14","arxiv_id":"2410.10665","n_code_links":1,"syntology":null},{"paper":"/paper/one-language-many-gaps-evaluating-dialect","slug":"one-language-many-gaps-evaluating-dialect","title":"One Language, Many Gaps: Evaluating Dialect Fairness and Robustness of Large Language Models in Reasoning Tasks","date":"2024-10-14","arxiv_id":"2410.11005","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["fangru-lin/redial_dialect_robustness_fairness"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"performance-in-a-dialectal-profiling-task-of","title":"Performance in a dialectal profiling task of LLMs for varieties of Brazilian Portuguese","date":"2024-10-14","arxiv_id":"2410.10991","n_code_links":0,"syntology":null},{"paper":"/paper/towards-better-multi-head-attention-via","slug":"towards-better-multi-head-attention-via","title":"Towards Better Multi-head Attention via Channel-wise Sample Permutation","date":"2024-10-14","arxiv_id":"2410.10914","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["dashenzi721/csp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-gender-bias-of-llms-in-making","title":"Evaluating Gender Bias of LLMs in Making Morality Judgements","date":"2024-10-13","arxiv_id":"2410.09992","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-implicit-bias-in-large-language","title":"Investigating Implicit Bias in Large Language Models: A Large-Scale Study of Over 50 LLMs","date":"2024-10-13","arxiv_id":"2410.12864","n_code_links":0,"syntology":null},{"paper":null,"slug":"m2m-gen-a-multimodal-framework-for-automated","title":"M2M-Gen: A Multimodal Framework for Automated Background Music Generation in Japanese Manga Using Large Language Models","date":"2024-10-13","arxiv_id":"2410.09928","n_code_links":0,"syntology":null},{"paper":null,"slug":"llinstruct-an-instruction-tuned-model-for","title":"\\llinstruct: An Instruction-tuned model for English Language Proficiency Assessments","date":"2024-10-12","arxiv_id":"2410.09314","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-in-house-large-language-models-to","title":"Fine-Tuning In-House Large Language Models to Infer Differential Diagnosis from Radiology Reports","date":"2024-10-11","arxiv_id":"2410.09234","n_code_links":0,"syntology":null},{"paper":null,"slug":"humanity-in-ai-detecting-the-personality-of","title":"Humanity in AI: Detecting the Personality of Large Language Models","date":"2024-10-11","arxiv_id":"2410.08545","n_code_links":0,"syntology":null},{"paper":"/paper/synth-sonar-sonar-image-synthesis-with","slug":"synth-sonar-sonar-image-synthesis-with","title":"Synth-SONAR: Sonar Image Synthesis with Enhanced Diversity and Realism via Dual Diffusion Models and GPT Prompting","date":"2024-10-11","arxiv_id":"2410.08612","n_code_links":1,"syntology":null},{"paper":null,"slug":"capturing-bias-diversity-in-llms","title":"Capturing Bias Diversity in LLMs","date":"2024-10-09","arxiv_id":"2410.12839","n_code_links":0,"syntology":null},{"paper":null,"slug":"auto-evolve-enhancing-large-language-model-s","title":"Auto-Evolve: Enhancing Large Language Model's Performance via Self-Reasoning Framework","date":"2024-10-08","arxiv_id":"2410.06328","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-free-energy-in-pretraining-model","title":"Leveraging free energy in pretraining model selection for improved fine-tuning","date":"2024-10-08","arxiv_id":"2410.05612","n_code_links":0,"syntology":null},{"paper":null,"slug":"anyattack-towards-large-scale-self-supervised","title":"AnyAttack: Towards Large-scale Self-supervised Adversarial Attacks on Vision-language Models","date":"2024-10-07","arxiv_id":"2410.05346","n_code_links":0,"syntology":null},{"paper":"/paper/famma-a-benchmark-for-financial-domain","slug":"famma-a-benchmark-for-financial-domain","title":"FAMMA: A Benchmark for Financial Domain Multilingual Multimodal Question Answering","date":"2024-10-06","arxiv_id":"2410.04526","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["famma-bench/bench-script"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-model-inference-acceleration-a","slug":"large-language-model-inference-acceleration-a","title":"Large Language Model Inference Acceleration: A Comprehensive Hardware Perspective","date":"2024-10-06","arxiv_id":"2410.04466","n_code_links":1,"syntology":null},{"paper":null,"slug":"protocollm-automatic-evaluation-framework-of","title":"ProtocoLLM: Automatic Evaluation Framework of LLMs on Domain-Specific Scientific Protocol Formulation Tasks","date":"2024-10-06","arxiv_id":"2410.04601","n_code_links":0,"syntology":null},{"paper":"/paper/gamified-crowd-sourcing-of-high-quality-data","slug":"gamified-crowd-sourcing-of-high-quality-data","title":"Gamified crowd-sourcing of high-quality data for visual fine-tuning","date":"2024-10-05","arxiv_id":"2410.04038","n_code_links":0,"syntology":null}],"record_sha256":"ae724071abd51a1ce43b567e2d4958c1cbb1c098ce30f63cfcba482996bbcd31","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}