{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-3/papers/3","list_of":"/method/gpt-3","method":"GPT-3","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":3,"pages_in_order":20,"rows_per_page":100,"rows":[201,300],"of":1906,"counts":{"archive_papers_tagged":1906,"with_a_code_link":866,"where_syntology_ran_a_sample":319,"not_listed_spam_title":0,"listed":1906,"listed_where_code_ran":319,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":259,"every_run_a_failure_of_syntologys_instrument":60,"listed_with_a_run_with_no_instrument_failure":259,"listed_every_run_a_failure_of_syntologys_instrument":60,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-3","prev":"/method/gpt-3/papers/2","next":"/method/gpt-3/papers/4","papers":[{"paper":"/paper/sbi-rag-enhancing-math-word-problem-solving","slug":"sbi-rag-enhancing-math-word-problem-solving","title":"SBI-RAG: Enhancing Math Word Problem Solving for Students through Schema-Based Instruction and Retrieval-Augmented Generation","date":"2024-10-17","arxiv_id":"2410.13293","n_code_links":1,"syntology":null},{"paper":"/paper/agent-skill-acquisition-for-large-language","slug":"agent-skill-acquisition-for-large-language","title":"Agent Skill Acquisition for Large Language Models via CycleQD","date":"2024-10-16","arxiv_id":"2410.14735","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["SakanaAI/CycleQD"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"table-llm-specialist-language-model","title":"Table-LLM-Specialist: Language Model Specialists for Tables using Iterative Generator-Validator Fine-tuning","date":"2024-10-16","arxiv_id":"2410.12164","n_code_links":0,"syntology":null},{"paper":null,"slug":"code-mixer-ya-nahi-novel-approaches-to","title":"Code-Mixer Ya Nahi: Novel Approaches to Measuring Multilingual LLMs' Code-Mixing Capabilities","date":"2024-10-14","arxiv_id":"2410.11079","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-legal-judgement-prediction-in-a","slug":"rethinking-legal-judgement-prediction-in-a","title":"Rethinking Legal Judgement Prediction in a Realistic Scenario in the Era of Large Language Models","date":"2024-10-14","arxiv_id":"2410.10542","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-gender-bias-of-llms-in-making","title":"Evaluating Gender Bias of LLMs in Making Morality Judgements","date":"2024-10-13","arxiv_id":"2410.09992","n_code_links":0,"syntology":null},{"paper":null,"slug":"llinstruct-an-instruction-tuned-model-for","title":"\\llinstruct: An Instruction-tuned model for English Language Proficiency Assessments","date":"2024-10-12","arxiv_id":"2410.09314","n_code_links":0,"syntology":null},{"paper":"/paper/attngcg-enhancing-jailbreaking-attacks-on","slug":"attngcg-enhancing-jailbreaking-attacks-on","title":"AttnGCG: Enhancing Jailbreaking Attacks on LLMs with Attention Manipulation","date":"2024-10-11","arxiv_id":"2410.09040","n_code_links":1,"syntology":null},{"paper":null,"slug":"observing-the-southern-us-culture-of-honor","title":"Observing the Southern US Culture of Honor Using Large-Scale Social Media Analysis","date":"2024-10-11","arxiv_id":"2410.13887","n_code_links":0,"syntology":null},{"paper":"/paper/socialgaze-improving-the-integration-of-human","slug":"socialgaze-improving-the-integration-of-human","title":"SocialGaze: Improving the Integration of Human Social Norms in Large Language Models","date":"2024-10-11","arxiv_id":"2410.08698","n_code_links":1,"syntology":null},{"paper":null,"slug":"flier-few-shot-language-image-models-embedded","title":"FLIER: Few-shot Language Image Models Embedded with Latent Representations","date":"2024-10-10","arxiv_id":"2410.07648","n_code_links":0,"syntology":null},{"paper":"/paper/the-rise-of-ai-generated-content-in-wikipedia","slug":"the-rise-of-ai-generated-content-in-wikipedia","title":"The Rise of AI-Generated Content in Wikipedia","date":"2024-10-10","arxiv_id":"2410.08044","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["brooksca3/wiki_collection"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"autofeedback-an-llm-based-framework-for","title":"AutoFeedback: An LLM-based Framework for Efficient and Accurate API Request Generation","date":"2024-10-09","arxiv_id":"2410.06943","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-model-for-less-resourced-language","title":"Generative Model for Less-Resourced Language with 1 billion parameters","date":"2024-10-09","arxiv_id":"2410.06898","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-as-code-executors-an","title":"Large Language Models as Code Executors: An Exploratory Study","date":"2024-10-09","arxiv_id":"2410.06667","n_code_links":0,"syntology":null},{"paper":"/paper/mentalarena-self-play-training-of-language","slug":"mentalarena-self-play-training-of-language","title":"MentalArena: Self-play Training of Language Models for Diagnosis and Treatment of Mental Health Disorders","date":"2024-10-09","arxiv_id":"2410.06845","n_code_links":1,"syntology":null},{"paper":"/paper/narrative-of-thought-improving-temporal","slug":"narrative-of-thought-improving-temporal","title":"Narrative-of-Thought: Improving Temporal Reasoning of Large Language Models via Recounted Narratives","date":"2024-10-07","arxiv_id":"2410.05558","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-instruction-finetuning-neural-machine","title":"On Instruction-Finetuning Neural Machine Translation Models","date":"2024-10-07","arxiv_id":"2410.05553","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-for-knowledge-free","title":"Large Language Models for Knowledge-Free Network Management: Feasibility Study and Opportunities","date":"2024-10-06","arxiv_id":"2410.17259","n_code_links":0,"syntology":null},{"paper":"/paper/take-it-easy-label-adaptive-self","slug":"take-it-easy-label-adaptive-self","title":"Take It Easy: Label-Adaptive Self-Rationalization for Fact Verification and Explanation Generation","date":"2024-10-05","arxiv_id":"2410.04002","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jingyng/label-adaptive-self-rationalization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"crafting-narrative-closures-zero-shot","title":"Crafting Narrative Closures: Zero-Shot Learning with SSM Mamba for Short Story Ending Generation","date":"2024-10-04","arxiv_id":"2410.10848","n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-lingual-transfer-for-automatic-question","title":"Cross-lingual Transfer for Automatic Question Generation by Learning Interrogative Structures in Target Languages","date":"2024-10-04","arxiv_id":"2410.03197","n_code_links":0,"syntology":null},{"paper":null,"slug":"structured-list-grounded-question-answering","title":"Structured List-Grounded Question Answering","date":"2024-10-04","arxiv_id":"2410.03950","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-linguistically-aware-and-language","title":"Towards Linguistically-Aware and Language-Independent Tokenization for Large Language Models (LLMs)","date":"2024-10-04","arxiv_id":"2410.03568","n_code_links":0,"syntology":null},{"paper":"/paper/codejudge-evaluating-code-generation-with","slug":"codejudge-evaluating-code-generation-with","title":"CodeJudge: Evaluating Code Generation with Large Language Models","date":"2024-10-03","arxiv_id":"2410.02184","n_code_links":1,"syntology":{"ran":15,"of":23,"n_ran_checked":14,"n_instrument":1,"unverified":8,"pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 2 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","official":{"repos":["VichyTong/CodeJudge"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":7,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/visual-editing-with-llm-based-tool-chaining","slug":"visual-editing-with-llm-based-tool-chaining","title":"Visual Editing with LLM-based Tool Chaining: An Efficient Distillation Approach for Real-Time Applications","date":"2024-10-03","arxiv_id":"2410.02952","n_code_links":1,"syntology":null},{"paper":null,"slug":"emotion-aware-response-generation-using","title":"Emotion-Aware Embedding Fusion in LLMs (Flan-T5, LLAMA 2, DeepSeek-R1, and ChatGPT 4) for Intelligent Response Generation","date":"2024-10-02","arxiv_id":"2410.01306","n_code_links":0,"syntology":null},{"paper":"/paper/alignsum-data-pyramid-hierarchical-fine","slug":"alignsum-data-pyramid-hierarchical-fine","title":"AlignSum: Data Pyramid Hierarchical Fine-tuning for Aligning with Human Summarization Preference","date":"2024-10-01","arxiv_id":"2410.00409","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["csyanghan/alignsum"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"decoding-hate-exploring-language-models","title":"Decoding Hate: Exploring Language Models' Reactions to Hate Speech","date":"2024-10-01","arxiv_id":"2410.00775","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-enhanced-model-for-eye-leme-an-open","title":"Language Enhanced Model for Eye (LEME): An Open-Source Ophthalmology-Specific Large Language Model","date":"2024-10-01","arxiv_id":"2410.03740","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-looming-replication-crisis-in-evaluating","title":"A Looming Replication Crisis in Evaluating Behavior in Language Models? Evidence and Solutions","date":"2024-09-30","arxiv_id":"2409.20303","n_code_links":0,"syntology":null},{"paper":null,"slug":"adapting-llms-for-the-medical-domain-in","title":"Adapting LLMs for the Medical Domain in Portuguese: A Study on Fine-Tuning and Model Evaluation","date":"2024-09-30","arxiv_id":"2410.00163","n_code_links":0,"syntology":null},{"paper":null,"slug":"charting-the-future-using-chart-question","title":"Charting the Future: Using Chart Question-Answering for Scalable Evaluation of LLM-Driven Data Visualizations","date":"2024-09-27","arxiv_id":"2409.18764","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-in-domain-question-answering-for","title":"Efficient In-Domain Question Answering for Resource-Constrained Environments","date":"2024-09-26","arxiv_id":"2409.17648","n_code_links":0,"syntology":null},{"paper":"/paper/maskllm-learnable-semi-structured-sparsity","slug":"maskllm-learnable-semi-structured-sparsity","title":"MaskLLM: Learnable Semi-Structured Sparsity for Large Language Models","date":"2024-09-26","arxiv_id":"2409.17481","n_code_links":1,"syntology":{"ran":5,"of":16,"n_ran_checked":5,"n_instrument":0,"unverified":11,"pointer_only":16,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","official":{"repos":["nvlabs/maskllm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":11,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-prompting-based-representation-learning","title":"A Prompting-Based Representation Learning Method for Recommendation with Large Language Models","date":"2024-09-25","arxiv_id":"2409.16674","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-llm-for-real-time-transcription-and","title":"Using LLM for Real-Time Transcription and Summarization of Doctor-Patient Interactions into ePuskesmas in Indonesia","date":"2024-09-25","arxiv_id":"2409.17054","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-can-be-cognitively-biased-an-exploratory","title":"AI Can Be Cognitively Biased: An Exploratory Study on Threshold Priming in LLM-Based Batch Relevance Assessment","date":"2024-09-24","arxiv_id":"2409.16022","n_code_links":0,"syntology":null},{"paper":"/paper/effectiveness-of-cross-linguistic-extraction","slug":"effectiveness-of-cross-linguistic-extraction","title":"Effectiveness of Cross-linguistic Extraction of Genetic Information using Generative Large Language Models","date":"2024-09-24","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"selection-of-prompt-engineering-techniques","title":"Selection of Prompt Engineering Techniques for Code Generation through Predicting Code Complexity","date":"2024-09-24","arxiv_id":"2409.16416","n_code_links":0,"syntology":null},{"paper":null,"slug":"synatra-turning-indirect-knowledge-into","title":"Synatra: Turning Indirect Knowledge into Direct Demonstrations for Digital Agents at Scale","date":"2024-09-24","arxiv_id":"2409.15637","n_code_links":0,"syntology":null},{"paper":null,"slug":"task-oriented-prompt-enhancement-via-script","title":"Task-oriented Prompt Enhancement via Script Generation","date":"2024-09-24","arxiv_id":"2409.16418","n_code_links":0,"syntology":null},{"paper":"/paper/effective-and-evasive-fuzz-testing-driven","slug":"effective-and-evasive-fuzz-testing-driven","title":"PAPILLON: Efficient and Stealthy Fuzz Testing-Powered Jailbreaks for LLMs","date":"2024-09-23","arxiv_id":"2409.14866","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aaFrostnova/Papillon"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gem-rag-graphical-eigen-memories-for","title":"GEM-RAG: Graphical Eigen Memories For Retrieval Augmented Generation","date":"2024-09-23","arxiv_id":"2409.15566","n_code_links":0,"syntology":null},{"paper":null,"slug":"location-is-key-leveraging-large-language","title":"Location is Key: Leveraging Large Language Model for Functional Bug Localization in Verilog","date":"2024-09-23","arxiv_id":"2409.15186","n_code_links":0,"syntology":null},{"paper":"/paper/can-pre-trained-language-models-generate","slug":"can-pre-trained-language-models-generate","title":"Can pre-trained language models generate titles for research papers?","date":"2024-09-22","arxiv_id":"2409.14602","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-the-quality-of-code-comments","title":"Evaluating the Quality of Code Comments Generated by Large Language Models for Novice Programmers","date":"2024-09-22","arxiv_id":"2409.14368","n_code_links":0,"syntology":null},{"paper":null,"slug":"proof-automation-with-large-language-models","title":"Proof Automation with Large Language Models","date":"2024-09-22","arxiv_id":"2409.14274","n_code_links":0,"syntology":null},{"paper":"/paper/2409-14175","slug":"2409-14175","title":"QMOS: Enhancing LLMs for Telecommunication with Question Masked loss and Option Shuffling","date":"2024-09-21","arxiv_id":"2409.14175","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-in-triples-for-llms-enhancing-table","title":"Knowledge in Triples for LLMs: Enhancing Table QA Accuracy with Semantic Extraction","date":"2024-09-21","arxiv_id":"2409.14192","n_code_links":0,"syntology":null},{"paper":"/paper/neural-symbolic-collaborative-distillation","slug":"neural-symbolic-collaborative-distillation","title":"Neural-Symbolic Collaborative Distillation: Advancing Small Language Models for Complex Reasoning Tasks","date":"2024-09-20","arxiv_id":"2409.13203","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":4,"n_instrument":2,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["xnhyacinth/nesycd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-tinybert-for-financial-sentiment-1","slug":"enhancing-tinybert-for-financial-sentiment-1","title":"Enhancing TinyBERT for Financial Sentiment Analysis Using GPT-Augmented FinBERT Distillation","date":"2024-09-19","arxiv_id":"2409.18999","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-effectiveness-of-llms-for-manual-test","title":"On the Effectiveness of LLMs for Manual Test Verifications","date":"2024-09-19","arxiv_id":"2409.12405","n_code_links":0,"syntology":null},{"paper":null,"slug":"retrieval-augmented-test-generation-how-far","title":"Retrieval-Augmented Test Generation: How Far Are We?","date":"2024-09-19","arxiv_id":"2409.12682","n_code_links":0,"syntology":null},{"paper":"/paper/magicore-multi-agent-iterative-coarse-to-fine","slug":"magicore-multi-agent-iterative-coarse-to-fine","title":"MAgICoRe: Multi-Agent, Iterative, Coarse-to-Fine Refinement for Reasoning","date":"2024-09-18","arxiv_id":"2409.12147","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dinobby/magicore"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/tart-an-open-source-tool-augmented-framework","slug":"tart-an-open-source-tool-augmented-framework","title":"TART: An Open-Source Tool-Augmented Framework for Explainable Table-based Reasoning","date":"2024-09-18","arxiv_id":"2409.11724","n_code_links":1,"syntology":null},{"paper":"/paper/small-language-models-can-outperform-humans","slug":"small-language-models-can-outperform-humans","title":"Small Language Models can Outperform Humans in Short Creative Writing: A Study Comparing SLMs with Humans and LLMs","date":"2024-09-17","arxiv_id":"2409.11547","n_code_links":1,"syntology":null},{"paper":"/paper/benchmarking-large-language-model-uncertainty","slug":"benchmarking-large-language-model-uncertainty","title":"Benchmarking Large Language Model Uncertainty for Prompt Optimization","date":"2024-09-16","arxiv_id":"2409.10044","n_code_links":1,"syntology":null},{"paper":null,"slug":"llm-der-a-named-entity-recognition-method","title":"LLM-DER:A Named Entity Recognition Method Based on Large Language Models for Chinese Coal Chemical Domain","date":"2024-09-16","arxiv_id":"2409.10077","n_code_links":0,"syntology":null},{"paper":"/paper/select-sql-self-correcting-ensemble-chain-of","slug":"select-sql-self-correcting-ensemble-chain-of","title":"SelECT-SQL: Self-correcting ensemble Chain-of-Thought for Text-to-SQL","date":"2024-09-16","arxiv_id":"2409.10007","n_code_links":1,"syntology":null},{"paper":null,"slug":"detection-made-easy-potentials-of-large","title":"Detection Made Easy: Potentials of Large Language Models for Solidity Vulnerabilities","date":"2024-09-15","arxiv_id":"2409.10574","n_code_links":0,"syntology":null},{"paper":null,"slug":"rethinkmcts-refining-erroneous-thoughts-in","title":"RethinkMCTS: Refining Erroneous Thoughts in Monte Carlo Tree Search for Code Generation","date":"2024-09-15","arxiv_id":"2409.09584","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-empirical-evaluation-of-using-chatgpt-to","title":"An empirical evaluation of using ChatGPT to summarize disputes for recommending similar labor and employment cases in Chinese","date":"2024-09-14","arxiv_id":"2409.09280","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-ingredient-substitution-using","title":"Optimizing Ingredient Substitution Using Large Language Models to Enhance Phytochemical Content in Recipes","date":"2024-09-13","arxiv_id":"2409.08792","n_code_links":0,"syntology":null},{"paper":"/paper/can-large-language-models-unlock-novel","slug":"can-large-language-models-unlock-novel","title":"Can Large Language Models Unlock Novel Scientific Research Ideas?","date":"2024-09-10","arxiv_id":"2409.06185","n_code_links":1,"syntology":{"ran":2,"of":11,"n_ran_checked":2,"n_instrument":0,"unverified":9,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","official":{"repos":["sandeep82945/future-idea-generation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":"/paper/2409-13727","slug":"2409-13727","title":"Classification performance and reproducibility of GPT-4 omni for information extraction from veterinary electronic health records","date":"2024-09-09","arxiv_id":"2409.13727","n_code_links":1,"syntology":null},{"paper":null,"slug":"elsevier-arena-human-evaluation-of-chemistry","title":"Elsevier Arena: Human Evaluation of Chemistry/Biology/Health Foundational Large Language Models","date":"2024-09-09","arxiv_id":"2409.05486","n_code_links":0,"syntology":null},{"paper":null,"slug":"fairhome-a-fair-housing-and-fair-lending","title":"FairHome: A Fair Housing and Fair Lending Dataset","date":"2024-09-09","arxiv_id":"2409.05990","n_code_links":0,"syntology":null},{"paper":null,"slug":"harmonic-reasoning-in-large-language-models","title":"Harmonic Reasoning in Large Language Models","date":"2024-09-09","arxiv_id":"2409.05521","n_code_links":0,"syntology":null},{"paper":null,"slug":"identifying-the-sources-of-ideological-bias","title":"Identifying the sources of ideological bias in GPT models through linguistic variation in output","date":"2024-09-09","arxiv_id":"2409.06043","n_code_links":0,"syntology":null},{"paper":null,"slug":"regression-with-large-language-models-for","title":"Regression with Large Language Models for Materials and Molecular Property Prediction","date":"2024-09-09","arxiv_id":"2409.06080","n_code_links":0,"syntology":null},{"paper":"/paper/vision-fused-attack-advancing-aggressive-and","slug":"vision-fused-attack-advancing-aggressive-and","title":"Vision-fused Attack: Advancing Aggressive and Stealthy Adversarial Text against Neural Machine Translation","date":"2024-09-08","arxiv_id":"2409.05021","n_code_links":1,"syntology":{"ran":0,"of":4,"n_ran_checked":0,"n_instrument":0,"unverified":4,"pointer_only":4,"phrase":"0 ran · 4 unverified","official":{"repos":["levelower/vfa"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"paper":null,"slug":"towards-safer-online-spaces-simulating-and","title":"Towards Safer Online Spaces: Simulating and Assessing Intervention Strategies for Eating Disorder Discussions","date":"2024-09-06","arxiv_id":"2409.04043","n_code_links":0,"syntology":null},{"paper":null,"slug":"materialbench-evaluating-college-level","title":"MaterialBENCH: Evaluating College-Level Materials Science Problem-Solving Abilities of Large Language Models","date":"2024-09-05","arxiv_id":"2409.03161","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-study-on-large-language-models","title":"A Comparative Study on Large Language Models for Log Parsing","date":"2024-09-04","arxiv_id":"2409.02474","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-privacy-savvy-are-large-language-models-a","title":"How Privacy-Savvy Are Large Language Models? A Case Study on Compliance and Privacy Technical Review","date":"2024-09-04","arxiv_id":"2409.02375","n_code_links":0,"syntology":null},{"paper":null,"slug":"irrelevant-alternatives-bias-large-language","title":"Irrelevant Alternatives Bias Large Language Model Hiring Decisions","date":"2024-09-04","arxiv_id":"2409.15299","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-as-efficient-reward","title":"Large Language Models as Efficient Reward Function Searchers for Custom-Environment Multi-Objective Reinforcement Learning","date":"2024-09-04","arxiv_id":"2409.02428","n_code_links":0,"syntology":null},{"paper":"/paper/more-is-more-addition-bias-in-large-language","slug":"more-is-more-addition-bias-in-large-language","title":"More is More: Addition Bias in Large Language Models","date":"2024-09-04","arxiv_id":"2409.02569","n_code_links":1,"syntology":null},{"paper":"/paper/self-judge-selective-instruction-following","slug":"self-judge-selective-instruction-following","title":"Self-Judge: Selective Instruction Following with Alignment Self-Evaluation","date":"2024-09-02","arxiv_id":"2409.00935","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nusnlp/Self-J"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-knowledge-infusion-for-explainable","title":"Deep Knowledge-Infusion For Explainable Depression Detection","date":"2024-09-01","arxiv_id":"2409.02122","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-large-language-models-address-open-target","title":"Can Large Language Models Address Open-Target Stance Detection?","date":"2024-08-30","arxiv_id":"2409.00222","n_code_links":0,"syntology":null},{"paper":"/paper/progres-prompted-generative-rescoring-on-asr","slug":"progres-prompted-generative-rescoring-on-asr","title":"ProGRes: Prompted Generative Rescoring on ASR n-Best","date":"2024-08-30","arxiv_id":"2409.00217","n_code_links":1,"syntology":null},{"paper":null,"slug":"fractured-sorry-bench-framework-for-revealing","title":"FRACTURED-SORRY-Bench: Framework for Revealing Attacks in Conversational Turns Undermining Refusal Efficacy and Defenses over SORRY-Bench (Automated Multi-shot Jailbreaks)","date":"2024-08-28","arxiv_id":"2408.16163","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-large-language-models-for-wireless","title":"Leveraging Large Language Models for Wireless Symbol Detection via In-Context Learning","date":"2024-08-28","arxiv_id":"2409.00124","n_code_links":0,"syntology":null},{"paper":null,"slug":"strategic-optimization-and-challenges-of","title":"Strategic Optimization and Challenges of Large Language Models in Object-Oriented Programming","date":"2024-08-27","arxiv_id":"2408.14834","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-visual-reasoning-by-vision-language","title":"Zero-Shot Visual Reasoning by Vision-Language Models: Benchmarking and Analysis","date":"2024-08-27","arxiv_id":"2409.00106","n_code_links":0,"syntology":null},{"paper":"/paper/codegraph-enhancing-graph-reasoning-of-llms","slug":"codegraph-enhancing-graph-reasoning-of-llms","title":"CodeGraph: Enhancing Graph Reasoning of LLMs with Code","date":"2024-08-25","arxiv_id":"2408.13863","n_code_links":1,"syntology":null},{"paper":"/paper/vision-language-and-large-language-model","slug":"vision-language-and-large-language-model","title":"Vision-Language and Large Language Model Performance in Gastroenterology: GPT, Claude, Llama, Phi, Mistral, Gemma, and Quantized Models","date":"2024-08-25","arxiv_id":"2409.00084","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-llm-based-automated-program-repair","title":"Enhancing Automated Program Repair with Solution Design","date":"2024-08-22","arxiv_id":"2408.12056","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-performance-how-compact-models","title":"Optimizing Performance: How Compact Models Match or Exceed GPT's Classification Capabilities through Fine-Tuning","date":"2024-08-22","arxiv_id":"2409.11408","n_code_links":0,"syntology":null},{"paper":"/paper/unlocking-adversarial-suffix-optimization","slug":"unlocking-adversarial-suffix-optimization","title":"Unlocking Adversarial Suffix Optimization Without Affirmative Phrases: Efficient Black-box Jailbreaking via LLM as Optimizer","date":"2024-08-21","arxiv_id":"2408.11313","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lenijwp/eclipse"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ctp-llm-clinical-trial-phase-transition","title":"CTP-LLM: Clinical Trial Phase Transition Prediction Using Large Language Models","date":"2024-08-20","arxiv_id":"2408.10995","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-well-do-large-language-models-serve-as","title":"How Well Do Large Language Models Serve as End-to-End Secure Code Agents for Python?","date":"2024-08-20","arxiv_id":"2408.10495","n_code_links":0,"syntology":null},{"paper":"/paper/soda-eval-open-domain-dialogue-evaluation-in","slug":"soda-eval-open-domain-dialogue-evaluation-in","title":"Soda-Eval: Open-Domain Dialogue Evaluation in the age of LLMs","date":"2024-08-20","arxiv_id":"2408.10902","n_code_links":1,"syntology":null},{"paper":null,"slug":"towardseffective-teaching-assistants-from","title":"Towardseffective teaching assistants: From intent-based chatbots to LLM-poweredteachingassistants","date":"2024-08-20","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"tablebench-a-comprehensive-and-complex","title":"TableBench: A Comprehensive and Complex Benchmark for Table Question Answering","date":"2024-08-17","arxiv_id":"2408.09174","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-mean-field-ansatz-for-zero-shot-weight","title":"A Mean Field Ansatz for Zero-Shot Weight Transfer","date":"2024-08-16","arxiv_id":"2408.08681","n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-llms-for-autonomous-spacecraft","slug":"fine-tuning-llms-for-autonomous-spacecraft","title":"Fine-tuning LLMs for Autonomous Spacecraft Control: A Case Study Using Kerbal Space Program","date":"2024-08-16","arxiv_id":"2408.08676","n_code_links":1,"syntology":null},{"paper":"/paper/fusechat-knowledge-fusion-of-chat-models-1","slug":"fusechat-knowledge-fusion-of-chat-models-1","title":"FuseChat: Knowledge Fusion of Chat Models","date":"2024-08-15","arxiv_id":"2408.07990","n_code_links":3,"syntology":null}],"record_sha256":"02a38e5d0c4e9771db9f8d28e1d4081817a36f068c724f068903001b57939eda","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}