{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-3/papers/4","list_of":"/method/gpt-3","method":"GPT-3","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":4,"pages_in_order":20,"rows_per_page":100,"rows":[301,400],"of":1906,"counts":{"archive_papers_tagged":1906,"with_a_code_link":866,"where_syntology_ran_a_sample":319,"not_listed_spam_title":0,"listed":1906,"listed_where_code_ran":319,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":259,"every_run_a_failure_of_syntologys_instrument":60,"listed_with_a_run_with_no_instrument_failure":259,"listed_every_run_a_failure_of_syntologys_instrument":60,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-3","prev":"/method/gpt-3/papers/3","next":"/method/gpt-3/papers/5","papers":[{"paper":"/paper/leveraging-web-crawled-data-for-high-quality","slug":"leveraging-web-crawled-data-for-high-quality","title":"Leveraging Web-Crawled Data for High-Quality Fine-Tuning","date":"2024-08-15","arxiv_id":"2408.08003","n_code_links":1,"syntology":null},{"paper":null,"slug":"predicting-lung-cancer-patient-prognosis-with","title":"Predicting Lung Cancer Patient Prognosis with Large Language Models","date":"2024-08-15","arxiv_id":"2408.07971","n_code_links":0,"syntology":null},{"paper":null,"slug":"codemirage-hallucinations-in-code-generated","title":"CodeMirage: Hallucinations in Code Generated by Large Language Models","date":"2024-08-14","arxiv_id":"2408.08333","n_code_links":0,"syntology":null},{"paper":null,"slug":"sage-rt-synthetic-alignment-data-generation","title":"SAGE-RT: Synthetic Alignment data Generation for Safety Evaluation and Red Teaming","date":"2024-08-14","arxiv_id":"2408.11851","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-cultural-adaptability-of-a-large","slug":"evaluating-cultural-adaptability-of-a-large","title":"Evaluating Cultural Adaptability of a Large Language Model via Simulation of Synthetic Personas","date":"2024-08-13","arxiv_id":"2408.06929","n_code_links":1,"syntology":null},{"paper":"/paper/kov-transferable-and-naturalistic-black-box","slug":"kov-transferable-and-naturalistic-black-box","title":"Kov: Transferable and Naturalistic Black-Box LLM Attacks using Markov Decision Processes and Tree Search","date":"2024-08-11","arxiv_id":"2408.08899","n_code_links":1,"syntology":null},{"paper":"/paper/utilizing-large-language-models-to-optimize","slug":"utilizing-large-language-models-to-optimize","title":"PhishLang: A Real-Time, Fully Client-Side Phishing Detection Framework Using MobileBERT","date":"2024-08-11","arxiv_id":"2408.05667","n_code_links":2,"syntology":null},{"paper":null,"slug":"chain-of-condition-construct-verify-and-solve","title":"Chain of Condition: Construct, Verify and Solve Conditions for Conditional Question Answering","date":"2024-08-10","arxiv_id":"2408.05442","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-the-code-debugging-ability-of-llms","slug":"enhancing-the-code-debugging-ability-of-llms","title":"COAST: Enhancing the Code Debugging Ability of LLMs through Communicative Agent Based Data Synthesis","date":"2024-08-09","arxiv_id":"2408.05006","n_code_links":1,"syntology":null},{"paper":null,"slug":"examining-the-behavior-of-llm-architectures","title":"Examining the Behavior of LLM Architectures Within the Framework of Standardized National Exams in Brazil","date":"2024-08-09","arxiv_id":"2408.05035","n_code_links":0,"syntology":null},{"paper":null,"slug":"could-chatgpt-get-an-engineering-degree","title":"Could ChatGPT get an Engineering Degree? Evaluating Higher Education Vulnerability to AI Assistants","date":"2024-08-07","arxiv_id":"2408.11841","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-llms-for-enhanced-open-vocabulary","slug":"leveraging-llms-for-enhanced-open-vocabulary","title":"Query3D: LLM-Powered Open-Vocabulary Scene Segmentation with Language Embedded 3D Gaussian","date":"2024-08-07","arxiv_id":"2408.03516","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-use-of-large-language-models-llm-for","title":"The Use of Large Language Models (LLM) for Cyber Threat Intelligence (CTI) in Cybercrime Forums","date":"2024-08-06","arxiv_id":"2408.03354","n_code_links":0,"syntology":null},{"paper":"/paper/2408-02416","slug":"2408-02416","title":"Why Are My Prompts Leaked? Unraveling Prompt Extraction Threats in Customized Large Language Models","date":"2024-08-05","arxiv_id":"2408.02416","n_code_links":1,"syntology":null},{"paper":"/paper/xmainframe-a-large-language-model-for","slug":"xmainframe-a-large-language-model-for","title":"XMainframe: A Large Language Model for Mainframe Modernization","date":"2024-08-05","arxiv_id":"2408.04660","n_code_links":1,"syntology":null},{"paper":null,"slug":"2408-02103","title":"Effective Demonstration Annotation for In-Context Learning via Language Model-Based Determinantal Point Process","date":"2024-08-04","arxiv_id":"2408.02103","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-large-language-models-with-chain","title":"Leveraging Large Language Models with Chain-of-Thought and Prompt Engineering for Traffic Crash Severity Analysis and Inference","date":"2024-08-04","arxiv_id":"2408.04652","n_code_links":0,"syntology":null},{"paper":null,"slug":"building-trust-in-mental-health-chatbots","title":"Building Trust in Mental Health Chatbots: Safety Metrics and LLM-Based Evaluation Tools","date":"2024-08-03","arxiv_id":"2408.04650","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-01055","title":"LLM as Runtime Error Handler: A Promising Pathway to Adaptive Self-Healing of Software Systems","date":"2024-08-02","arxiv_id":"2408.01055","n_code_links":0,"syntology":null},{"paper":null,"slug":"2408-01214","title":"High-Throughput Phenotyping of Clinical Text Using Large Language Models","date":"2024-08-02","arxiv_id":"2408.01214","n_code_links":0,"syntology":null},{"paper":"/paper/2408-00727","slug":"2408-00727","title":"Improving Retrieval-Augmented Generation in Medicine with Iterative Follow-up Questions","date":"2024-08-01","arxiv_id":"2408.00727","n_code_links":1,"syntology":null},{"paper":"/paper/2408-00764","slug":"2408-00764","title":"AgentGen: Enhancing Planning Abilities for Large Language Model based Agent via Environment and Task Generation","date":"2024-08-01","arxiv_id":"2408.00764","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["lazychih114/AgentGen-Reproduction"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"2407-21443","title":"Improving Faithfulness of Large Language Models in Summarization via Sliding Generation and Self-Consistency","date":"2024-07-31","arxiv_id":"2407.21443","n_code_links":0,"syntology":null},{"paper":"/paper/2407-21170","slug":"2407-21170","title":"Decomposed Prompting to Answer Questions on a Course Discussion Board","date":"2024-07-30","arxiv_id":"2407.21170","n_code_links":1,"syntology":null},{"paper":null,"slug":"breaking-agents-compromising-autonomous-llm","title":"Breaking Agents: Compromising Autonomous LLM Agents Through Malfunction Amplification","date":"2024-07-30","arxiv_id":"2407.20859","n_code_links":0,"syntology":null},{"paper":"/paper/comparison-of-large-language-models-for","slug":"comparison-of-large-language-models-for","title":"Comparison of Large Language Models for Generating Contextually Relevant Questions","date":"2024-07-30","arxiv_id":"2407.20578","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-code-translation-in-language-models","title":"Enhancing Code Translation in Language Models with Few-Shot Learning via Retrieval-Augmented Generation","date":"2024-07-29","arxiv_id":"2407.19619","n_code_links":0,"syntology":null},{"paper":"/paper/are-llms-good-annotators-for-discourse-level","slug":"are-llms-good-annotators-for-discourse-level","title":"Are LLMs Good Annotators for Discourse-level Event Relation Extraction?","date":"2024-07-28","arxiv_id":"2407.19568","n_code_links":1,"syntology":null},{"paper":null,"slug":"tagify-llm-powered-tagging-interface-for","title":"TAGIFY: LLM-powered Tagging Interface for Improved Data Findability on OGD portals","date":"2024-07-26","arxiv_id":"2407.18764","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-large-language-models-for-the","title":"Using Large Language Models for the Interpretation of Building Regulations","date":"2024-07-26","arxiv_id":"2407.21060","n_code_links":0,"syntology":null},{"paper":null,"slug":"closing-the-gap-between-open-source-and","title":"Closing the gap between open-source and commercial large language models for medical evidence summarization","date":"2024-07-25","arxiv_id":"2408.00588","n_code_links":0,"syntology":null},{"paper":"/paper/cost-effective-instruction-learning-for","slug":"cost-effective-instruction-learning-for","title":"Cost-effective Instruction Learning for Pathology Vision and Language Analysis","date":"2024-07-25","arxiv_id":"2407.17734","n_code_links":1,"syntology":null},{"paper":"/paper/peft-u-parameter-efficient-fine-tuning-for","slug":"peft-u-parameter-efficient-fine-tuning-for","title":"PEFT-U: Parameter-Efficient Fine-Tuning for User Personalization","date":"2024-07-25","arxiv_id":"2407.18078","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ChrisIsKing/Parameter-Efficient-Personalization"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"bailicai-a-domain-optimized-retrieval","title":"Bailicai: A Domain-Optimized Retrieval-Augmented Generation Framework for Medical Applications","date":"2024-07-24","arxiv_id":"2407.21055","n_code_links":0,"syntology":null},{"paper":null,"slug":"testing-large-language-models-on-driving","title":"Testing Large Language Models on Driving Theory Knowledge and Skills for Connected Autonomous Vehicles","date":"2024-07-24","arxiv_id":"2407.17211","n_code_links":0,"syntology":null},{"paper":"/paper/data-mixture-inference-what-do-bpe-tokenizers","slug":"data-mixture-inference-what-do-bpe-tokenizers","title":"Data Mixture Inference: What do BPE Tokenizers Reveal about their Training Data?","date":"2024-07-23","arxiv_id":"2407.16607","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alisawuffles/tokenizer-attack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-llm-s-cognition-via-structurization","slug":"enhancing-llm-s-cognition-via-structurization","title":"Enhancing LLM's Cognition via Structurization","date":"2024-07-23","arxiv_id":"2407.16434","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["alibaba/struxgpt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/patched-rtc-evaluating-llms-for-diverse","slug":"patched-rtc-evaluating-llms-for-diverse","title":"Patched RTC: evaluating LLMs for diverse software development tasks","date":"2024-07-23","arxiv_id":"2407.16557","n_code_links":1,"syntology":null},{"paper":"/paper/robust-privacy-amidst-innovation-with-large","slug":"robust-privacy-amidst-innovation-with-large","title":"Robust Privacy Amidst Innovation with Large Language Models Through a Critical Assessment of the Risks","date":"2024-07-23","arxiv_id":"2407.16166","n_code_links":1,"syntology":null},{"paper":null,"slug":"imposter-ai-adversarial-attacks-with-hidden","title":"Imposter.AI: Adversarial Attacks with Hidden Intentions towards Aligned Large Language Models","date":"2024-07-22","arxiv_id":"2407.15399","n_code_links":0,"syntology":null},{"paper":"/paper/mminstruct-a-high-quality-multi-modal","slug":"mminstruct-a-high-quality-multi-modal","title":"MMInstruct: A High-Quality Multi-Modal Instruction Tuning Dataset with Extensive Diversity","date":"2024-07-22","arxiv_id":"2407.15838","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yuecao0119/mminstruct"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/radiorag-factual-large-language-models-for","slug":"radiorag-factual-large-language-models-for","title":"RadioRAG: Factual large language models for enhanced diagnostics in radiology using online retrieval augmented generation","date":"2024-07-22","arxiv_id":"2407.15621","n_code_links":1,"syntology":null},{"paper":null,"slug":"unlocking-the-potential-benchmarking-large","title":"Unlocking the Potential: Benchmarking Large Language Models in Water Engineering and Research","date":"2024-07-22","arxiv_id":"2407.21045","n_code_links":0,"syntology":null},{"paper":null,"slug":"sqlfuse-enhancing-text-to-sql-performance","title":"SQLfuse: Enhancing Text-to-SQL Performance through Comprehensive LLM Synergy","date":"2024-07-19","arxiv_id":"2407.14568","n_code_links":0,"syntology":null},{"paper":"/paper/can-open-source-llms-compete-with-commercial","slug":"can-open-source-llms-compete-with-commercial","title":"Can Open-Source LLMs Compete with Commercial Models? Exploring the Few-Shot Performance of Current GPT Models in Biomedical Tasks","date":"2024-07-18","arxiv_id":"2407.13511","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-as-reliable-knowledge","title":"How Reliable are LLMs as Knowledge Bases? Re-thinking Facutality and Consistency","date":"2024-07-18","arxiv_id":"2407.13578","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-from-mistakes-prompting-for","title":"Learning-From-Mistakes Prompting for Indigenous Language Translation","date":"2024-07-18","arxiv_id":"2407.13343","n_code_links":0,"syntology":null},{"paper":null,"slug":"pragyan-connecting-the-dots-in-tweets","title":"PRAGyan -- Connecting the Dots in Tweets","date":"2024-07-18","arxiv_id":"2407.13909","n_code_links":0,"syntology":null},{"paper":"/paper/does-refusal-training-in-llms-generalize-to","slug":"does-refusal-training-in-llms-generalize-to","title":"Does Refusal Training in LLMs Generalize to the Past Tense?","date":"2024-07-16","arxiv_id":"2407.11969","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tml-epfl/llm-past-tense"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gpt-assisted-annotation-of-rhetorical-and","title":"GPT Assisted Annotation of Rhetorical and Linguistic Features for Interpretable Propaganda Technique Detection in News Text","date":"2024-07-16","arxiv_id":"2407.11827","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-as-misleading","title":"Large Language Models as Misleading Assistants in Conversation","date":"2024-07-16","arxiv_id":"2407.11789","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-visual-language-models-are-also-good","title":"Large Visual-Language Models Are Also Good Classifiers: A Study of In-Context Multimodal Fake News Detection","date":"2024-07-16","arxiv_id":"2407.12879","n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-bias-in-political-sample","title":"Representation Bias in Political Sample Simulations with Large Language Models","date":"2024-07-16","arxiv_id":"2407.11409","n_code_links":0,"syntology":null},{"paper":null,"slug":"review-feedback-reason-refer-a-novel","title":"ReFeR: Improving Evaluation and Reasoning through Hierarchy of Models","date":"2024-07-16","arxiv_id":"2407.12877","n_code_links":0,"syntology":null},{"paper":null,"slug":"empowering-llms-for-verilog-generation","title":"CodeV: Empowering LLMs with HDL Generation through Multi-Level Summarization","date":"2024-07-15","arxiv_id":"2407.10424","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-llm-respondents-for-item","title":"Leveraging LLM-Respondents for Item Evaluation: a Psychometric Analysis","date":"2024-07-15","arxiv_id":"2407.10899","n_code_links":0,"syntology":null},{"paper":"/paper/think-on-graph-2-0-deep-and-interpretable","slug":"think-on-graph-2-0-deep-and-interpretable","title":"Think-on-Graph 2.0: Deep and Faithful Large Language Model Reasoning with Knowledge-guided Retrieval Augmented Generation","date":"2024-07-15","arxiv_id":"2407.10805","n_code_links":1,"syntology":{"ran":13,"of":16,"n_ran_checked":13,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["idea-finai/tog-2"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"self-evolving-gpt-a-lifelong-autonomous","title":"Self-Evolving GPT: A Lifelong Autonomous Experiential Learner","date":"2024-07-12","arxiv_id":"2407.08937","n_code_links":0,"syntology":null},{"paper":"/paper/show-don-t-tell-evaluating-large-language","slug":"show-don-t-tell-evaluating-large-language","title":"Show, Don't Tell: Evaluating Large Language Models Beyond Textual Understanding with ChildPlay","date":"2024-07-12","arxiv_id":"2407.11068","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-two-sides-of-the-coin-hallucination","title":"The Two Sides of the Coin: Hallucination Generation and Detection with LLMs as Evaluators for LLMs","date":"2024-07-12","arxiv_id":"2407.09152","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-is-judged-more-human-than-humans-in","title":"GPT-4 is judged more human than humans in displaced and inverted Turing tests","date":"2024-07-11","arxiv_id":"2407.08853","n_code_links":0,"syntology":null},{"paper":"/paper/llms-morphological-analyses-of-complex-fst","slug":"llms-morphological-analyses-of-complex-fst","title":"LLMs' morphological analyses of complex FST-generated Finnish words","date":"2024-07-11","arxiv_id":"2407.08269","n_code_links":1,"syntology":null},{"paper":"/paper/vox-populi-vox-ai-using-language-models-to","slug":"vox-populi-vox-ai-using-language-models-to","title":"Vox Populi, Vox AI? Using Language Models to Estimate German Public Opinion","date":"2024-07-11","arxiv_id":"2407.08563","n_code_links":1,"syntology":null},{"paper":"/paper/fsponer-few-shot-prompt-optimization-for","slug":"fsponer-few-shot-prompt-optimization-for","title":"FsPONER: Few-shot Prompt Optimization for Named Entity Recognition in Domain-specific Scenarios","date":"2024-07-10","arxiv_id":"2407.08035","n_code_links":1,"syntology":null},{"paper":null,"slug":"mixsumm-topic-based-data-augmentation-using","title":"A Guide To Effectively Leveraging LLMs for Low-Resource Text Summarization: Data Augmentation and Semi-supervised Approaches","date":"2024-07-10","arxiv_id":"2407.07341","n_code_links":0,"syntology":null},{"paper":null,"slug":"multilingual-blending-llm-safety-alignment","title":"Multilingual Blending: LLM Safety Alignment Evaluation with Language Mixture","date":"2024-07-10","arxiv_id":"2407.07342","n_code_links":0,"syntology":null},{"paper":"/paper/ai-ai-bias-large-language-models-favor-their","slug":"ai-ai-bias-large-language-models-favor-their","title":"AI AI Bias: Large Language Models Favor Their Own Generated Content","date":"2024-07-09","arxiv_id":"2407.12856","n_code_links":1,"syntology":null},{"paper":"/paper/chatgpt-doesn-t-trust-chargers-fans-guardrail","slug":"chatgpt-doesn-t-trust-chargers-fans-guardrail","title":"ChatGPT Doesn't Trust Chargers Fans: Guardrail Sensitivity in Context","date":"2024-07-09","arxiv_id":"2407.06866","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["vli31/llm-guardrail-sensitivity"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"identification-of-emotions-on-twitter-during","title":"Identification of emotions on Twitter during the 2022 electoral process in Colombia","date":"2024-07-09","arxiv_id":"2407.07258","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompting-techniques-for-secure-code","title":"Prompting Techniques for Secure Code Generation: A Systematic Investigation","date":"2024-07-09","arxiv_id":"2407.07064","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-large-language-models-for-generating","title":"Using Large Language Models for Generating Smart Contracts for Health Insurance from Textual Policies","date":"2024-07-09","arxiv_id":"2407.07019","n_code_links":0,"syntology":null},{"paper":"/paper/inversecoder-unleashing-the-power-of","slug":"inversecoder-unleashing-the-power-of","title":"InverseCoder: Self-improving Instruction-Tuned Code LLMs with Inverse-Instruct","date":"2024-07-08","arxiv_id":"2407.05700","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wyt2000/InverseCoder"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-understand-layouts","slug":"large-language-models-understand-layouts","title":"Large Language Models Understand Layout","date":"2024-07-08","arxiv_id":"2407.05750","n_code_links":1,"syntology":null},{"paper":"/paper/shine-saliency-aware-hierarchical-negative","slug":"shine-saliency-aware-hierarchical-negative","title":"SHINE: Saliency-aware HIerarchical NEgative Ranking for Compositional Temporal Grounding","date":"2024-07-06","arxiv_id":"2407.05118","n_code_links":1,"syntology":{"ran":11,"of":13,"n_ran_checked":10,"n_instrument":1,"unverified":2,"pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zxccade/shine"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"are-large-language-models-strategic-decision","title":"Are Large Language Models Strategic Decision Makers? A Study of Performance and Bias in Two-Player Non-Zero-Sum Games","date":"2024-07-05","arxiv_id":"2407.04467","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-data-to-commonsense-reasoning-the-use-of","title":"From Data to Commonsense Reasoning: The Use of Large Language Models for Explainable AI","date":"2024-07-04","arxiv_id":"2407.03778","n_code_links":0,"syntology":null},{"paper":null,"slug":"nutribench-a-dataset-for-evaluating-large","title":"NutriBench: A Dataset for Evaluating Large Language Models on Nutrition Estimation from Meal Descriptions","date":"2024-07-04","arxiv_id":"2407.12843","n_code_links":0,"syntology":null},{"paper":"/paper/planning-with-large-language-models-for","slug":"planning-with-large-language-models-for","title":"Controllable Conversations: Planning-Based Dialogue Agent with Large Language Models","date":"2024-07-04","arxiv_id":"2407.03884","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-automating-text-annotation-a-case","title":"Towards Automating Text Annotation: A Case Study on Semantic Proximity Annotation using GPT-4","date":"2024-07-04","arxiv_id":"2407.04130","n_code_links":0,"syntology":null},{"paper":null,"slug":"agentinstruct-toward-generative-teaching-with","title":"AgentInstruct: Toward Generative Teaching with Agentic Flows","date":"2024-07-03","arxiv_id":"2407.03502","n_code_links":0,"syntology":null},{"paper":null,"slug":"regurgitative-training-the-value-of-real-data","title":"Regurgitative Training: The Value of Real Data in Training Large Language Models","date":"2024-07-03","arxiv_id":"2407.12835","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-the-code-clone-detection-capability","title":"Assessing the Code Clone Detection Capability of Large Language Models","date":"2024-07-02","arxiv_id":"2407.02402","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-numeric-awards-in-context-dueling","title":"Beyond Numeric Awards: In-Context Dueling Bandits with LLM Agents","date":"2024-07-02","arxiv_id":"2407.01887","n_code_links":0,"syntology":null},{"paper":null,"slug":"grasp-a-grid-based-benchmark-for-evaluating","title":"GRASP: A Grid-Based Benchmark for Evaluating Commonsense Spatial Reasoning","date":"2024-07-02","arxiv_id":"2407.01892","n_code_links":0,"syntology":null},{"paper":"/paper/integrate-the-essence-and-eliminate-the-dross","slug":"integrate-the-essence-and-eliminate-the-dross","title":"Integrate the Essence and Eliminate the Dross: Fine-Grained Self-Consistency for Free-Form Language Generation","date":"2024-07-02","arxiv_id":"2407.02056","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["WangXinglin/FSC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/multilingual-trolley-problems-for-language","slug":"multilingual-trolley-problems-for-language","title":"Language Model Alignment in Multilingual Trolley Problems","date":"2024-07-02","arxiv_id":"2407.02273","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["causalNLP/moralmachine","causalnlp/multitp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/sop-unlock-the-power-of-social-facilitation","slug":"sop-unlock-the-power-of-social-facilitation","title":"SeqAR: Jailbreak LLMs with Sequential Auto-Generated Characters","date":"2024-07-02","arxiv_id":"2407.01902","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yang-yan-yang-yan/sop"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/increasing-model-capacity-for-free-a-simple","slug":"increasing-model-capacity-for-free-a-simple","title":"Increasing Model Capacity for Free: A Simple Strategy for Parameter Efficient Fine-tuning","date":"2024-07-01","arxiv_id":"2407.01320","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":6,"n_instrument":1,"unverified":3,"pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["lins-lab/capaboost"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"predicting-dc-link-capacitor-current-ripple","title":"Predicting DC-Link Capacitor Current Ripple in AC-DC Rectifier Circuits Using Fine-Tuned Large Language Models","date":"2024-07-01","arxiv_id":"2407.01724","n_code_links":0,"syntology":null},{"paper":null,"slug":"applying-rlaif-for-code-generation-with-api","title":"Applying RLAIF for Code Generation with API-usage in Lightweight LLMs","date":"2024-06-28","arxiv_id":"2406.20060","n_code_links":0,"syntology":null},{"paper":null,"slug":"fred-flexible-reduction-distribution","title":"FRED: Flexible REduction-Distribution Interconnect and Communication Implementation for Wafer-Scale Distributed Training of DNN Models","date":"2024-06-28","arxiv_id":"2406.19580","n_code_links":0,"syntology":null},{"paper":"/paper/shortcutsbench-a-large-scale-real-world","slug":"shortcutsbench-a-large-scale-real-world","title":"ShortcutsBench: A Large-Scale Real-world Benchmark for API-based Agents","date":"2024-06-28","arxiv_id":"2407.00132","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":4,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["eachsheep/shortcutsbench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/from-artificial-needles-to-real-haystacks","slug":"from-artificial-needles-to-real-haystacks","title":"From Artificial Needles to Real Haystacks: Improving Retrieval Capabilities in LLMs by Finetuning on Synthetic Data","date":"2024-06-27","arxiv_id":"2406.19292","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-model-arena-for-cross-lingual-sentiment","title":"The Model Arena for Cross-lingual Sentiment Analysis: A Comparative Study in the Era of Large Language Models","date":"2024-06-27","arxiv_id":"2406.19358","n_code_links":0,"syntology":null},{"paper":null,"slug":"apigen-automated-pipeline-for-generating","title":"APIGen: Automated Pipeline for Generating Verifiable and Diverse Function-Calling Datasets","date":"2024-06-26","arxiv_id":"2406.18518","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-entity-recognition-using-ensembles","title":"Improving Entity Recognition Using Ensembles of Deep Learning and Fine-tuned Large Language Models: A Case Study on Adverse Event Extraction from Multiple Sources","date":"2024-06-26","arxiv_id":"2406.18049","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-instruction-following-ability","title":"Evaluation of Instruction-Following Ability for Large Language Models on Story-Ending Generation","date":"2024-06-24","arxiv_id":"2406.16356","n_code_links":0,"syntology":null},{"paper":null,"slug":"plagbench-exploring-the-duality-of-large","title":"PlagBench: Exploring the Duality of Large Language Models in Plagiarism Generation and Detection","date":"2024-06-24","arxiv_id":"2406.16288","n_code_links":0,"syntology":null},{"paper":"/paper/the-gpt-writingprompts-dataset-a-comparative","slug":"the-gpt-writingprompts-dataset-a-comparative","title":"The GPT-WritingPrompts Dataset: A Comparative Analysis of Character Portrayal in Short Stories","date":"2024-06-24","arxiv_id":"2406.16767","n_code_links":1,"syntology":null},{"paper":"/paper/towards-better-graph-based-cross-document","slug":"towards-better-graph-based-cross-document","title":"Towards Better Graph-based Cross-document Relation Extraction via Non-bridge Entity Enhancement and Prediction Debiasing","date":"2024-06-24","arxiv_id":"2406.16529","n_code_links":1,"syntology":null}],"record_sha256":"e867dedeab57b90a465a03e88c722033f6e11974c3cb8a5bdec8662c2dcc52e7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}