{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-3/papers/9","list_of":"/method/gpt-3","method":"GPT-3","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":9,"pages_in_order":20,"rows_per_page":100,"rows":[801,900],"of":1906,"counts":{"archive_papers_tagged":1906,"with_a_code_link":866,"where_syntology_ran_a_sample":319,"not_listed_spam_title":0,"listed":1906,"listed_where_code_ran":319,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":259,"every_run_a_failure_of_syntologys_instrument":60,"listed_with_a_run_with_no_instrument_failure":259,"listed_every_run_a_failure_of_syntologys_instrument":60,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-3","prev":"/method/gpt-3/papers/8","next":"/method/gpt-3/papers/10","papers":[{"paper":null,"slug":"improving-classification-performance-with","title":"Improving Classification Performance With Human Feedback: Label a few, we label the rest","date":"2024-01-17","arxiv_id":"2401.09555","n_code_links":0,"syntology":null},{"paper":null,"slug":"application-of-llm-agents-in-recruitment-a","title":"Application of LLM Agents in Recruitment: A Novel Framework for Resume Screening","date":"2024-01-16","arxiv_id":"2401.08315","n_code_links":0,"syntology":null},{"paper":null,"slug":"rag-vs-fine-tuning-pipelines-tradeoffs-and-a","title":"RAG vs Fine-tuning: Pipelines, Tradeoffs, and a Case Study on Agriculture","date":"2024-01-16","arxiv_id":"2401.08406","n_code_links":0,"syntology":null},{"paper":"/paper/tuning-language-models-by-proxy","slug":"tuning-language-models-by-proxy","title":"Tuning Language Models by Proxy","date":"2024-01-16","arxiv_id":"2401.08565","n_code_links":2,"syntology":{"ran":9,"of":10,"n_ran_checked":7,"n_instrument":2,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["alisawuffles/proxy-tuning"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/harnessing-large-language-models-over","slug":"harnessing-large-language-models-over","title":"Harnessing Large Language Models Over Transformer Models for Detecting Bengali Depressive Social Media Text: A Comprehensive Study","date":"2024-01-14","arxiv_id":"2401.07310","n_code_links":1,"syntology":null},{"paper":null,"slug":"assessing-large-language-models-in-mechanical","title":"Assessing Large Language Models in Mechanical Engineering Education: A Study on Mechanics-Focused Conceptual Understanding","date":"2024-01-13","arxiv_id":"2401.12983","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-gpt-4-and-open-source-language","title":"Comparing GPT-4 and Open-Source Language Models in Misinformation Mitigation","date":"2024-01-12","arxiv_id":"2401.06920","n_code_links":0,"syntology":null},{"paper":"/paper/from-automation-to-augmentation-large","slug":"from-automation-to-augmentation-large","title":"Human-AI Collaborative Essay Scoring: A Dual-Process Framework with LLMs","date":"2024-01-12","arxiv_id":"2401.06431","n_code_links":1,"syntology":null},{"paper":"/paper/how-johnny-can-persuade-llms-to-jailbreak","slug":"how-johnny-can-persuade-llms-to-jailbreak","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","date":"2024-01-12","arxiv_id":"2401.06373","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chats-lab/persuasive_jailbreaker"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/intention-analysis-prompting-makes-large","slug":"intention-analysis-prompting-makes-large","title":"Intention Analysis Makes LLMs A Good Jailbreak Defender","date":"2024-01-12","arxiv_id":"2401.06561","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["alphadl/safellm_with_intentionanalysis"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"persianmind-a-cross-lingual-persian-english","title":"PersianMind: A Cross-Lingual Persian-English Large Language Model","date":"2024-01-12","arxiv_id":"2401.06466","n_code_links":0,"syntology":null},{"paper":"/paper/pizzacommonsense-learning-to-model","slug":"pizzacommonsense-learning-to-model","title":"PizzaCommonSense: Learning to Model Commonsense Reasoning about Intermediate Steps in Cooking Recipes","date":"2024-01-12","arxiv_id":"2401.06930","n_code_links":1,"syntology":null},{"paper":null,"slug":"mutation-based-consistency-testing-for","title":"Mutation-based Consistency Testing for Evaluating the Code Understanding Capability of LLMs","date":"2024-01-11","arxiv_id":"2401.05940","n_code_links":0,"syntology":null},{"paper":"/paper/the-benefits-of-a-concise-chain-of-thought-on","slug":"the-benefits-of-a-concise-chain-of-thought-on","title":"The Benefits of a Concise Chain of Thought on Problem-Solving in Large Language Models","date":"2024-01-11","arxiv_id":"2401.05618","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["matthewrenze/jhu-concise-cot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/autoact-automatic-agent-learning-from-scratch","slug":"autoact-automatic-agent-learning-from-scratch","title":"AutoAct: Automatic Agent Learning from Scratch for QA via Self-Planning","date":"2024-01-10","arxiv_id":"2401.05268","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":6,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zjunlp/autoact"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/infiagent-dabench-evaluating-agents-on-data","slug":"infiagent-dabench-evaluating-agents-on-data","title":"InfiAgent-DABench: Evaluating Agents on Data Analysis Tasks","date":"2024-01-10","arxiv_id":"2401.05507","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["infiagent/infiagent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"machine-teaching-for-building-modular-ai","title":"Can Active Label Correction Improve LLM-based Modular AI Systems?","date":"2024-01-10","arxiv_id":"2401.05467","n_code_links":0,"syntology":null},{"paper":null,"slug":"distortions-in-judged-spatial-relations-in","title":"Distortions in Judged Spatial Relations in Large Language Models","date":"2024-01-08","arxiv_id":"2401.04218","n_code_links":0,"syntology":null},{"paper":"/paper/llm4plc-harnessing-large-language-models-for","slug":"llm4plc-harnessing-large-language-models-for","title":"LLM4PLC: Harnessing Large Language Models for Verifiable Programming of PLCs in Industrial Control Systems","date":"2024-01-08","arxiv_id":"2401.05443","n_code_links":1,"syntology":null},{"paper":"/paper/mixtral-of-experts","slug":"mixtral-of-experts","title":"Mixtral of Experts","date":"2024-01-08","arxiv_id":"2401.04088","n_code_links":6,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","official":null}},{"paper":null,"slug":"d-causal-exploring-defeasibility-in-causal","title":"Exploring Defeasibility in Causal Reasoning","date":"2024-01-06","arxiv_id":"2401.03183","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-large-language-models-to-assess-tutors","title":"Using Large Language Models to Assess Tutors' Performance in Reacting to Students Making Math Errors","date":"2024-01-06","arxiv_id":"2401.03238","n_code_links":0,"syntology":null},{"paper":"/paper/deepseek-llm-scaling-open-source-language","slug":"deepseek-llm-scaling-open-source-language","title":"DeepSeek LLM: Scaling Open-Source Language Models with Longtermism","date":"2024-01-05","arxiv_id":"2401.02954","n_code_links":1,"syntology":null},{"paper":"/paper/parameter-efficient-sparsity-crafting-from","slug":"parameter-efficient-sparsity-crafting-from","title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","date":"2024-01-05","arxiv_id":"2401.02731","n_code_links":2,"syntology":null},{"paper":null,"slug":"re-evaluating-the-memory-balanced-pipeline","title":"Re-evaluating the Memory-balanced Pipeline Parallelism: BPipe","date":"2024-01-04","arxiv_id":"2401.02088","n_code_links":0,"syntology":null},{"paper":"/paper/vietnamese-poem-generation-the-prospect-of","slug":"vietnamese-poem-generation-the-prospect-of","title":"Vietnamese Poem Generation & The Prospect Of Cross-Language Poem-To-Poem Translation","date":"2024-01-02","arxiv_id":"2401.01078","n_code_links":1,"syntology":null},{"paper":"/paper/a-b-b-a-triggering-logical-reasoning-failures","slug":"a-b-b-a-triggering-logical-reasoning-failures","title":"LogicAsker: Evaluating and Improving the Logical Reasoning Ability of Large Language Models","date":"2024-01-01","arxiv_id":"2401.00757","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yxwan123/logicasker"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-language-models-aren-t-all-that-you","title":"Large Language Models aren't all that you need","date":"2024-01-01","arxiv_id":"2401.00698","n_code_links":0,"syntology":null},{"paper":"/paper/advancing-ttp-analysis-harnessing-the-power","slug":"advancing-ttp-analysis-harnessing-the-power","title":"Advancing TTP Analysis: Harnessing the Power of Large Language Models with Retrieval Augmented Generation","date":"2023-12-30","arxiv_id":"2401.00280","n_code_links":1,"syntology":null},{"paper":"/paper/jatmo-prompt-injection-defense-by-task","slug":"jatmo-prompt-injection-defense-by-task","title":"Jatmo: Prompt Injection Defense by Task-Specific Finetuning","date":"2023-12-29","arxiv_id":"2312.17673","n_code_links":1,"syntology":{"ran":10,"of":18,"n_ran_checked":10,"n_instrument":0,"unverified":8,"pointer_only":18,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","official":{"repos":["wagner-group/prompt-injection-defense"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-the-performance-of-large-language-1","title":"Evaluating the Performance of Large Language Models for Spanish Language in Undergraduate Admissions Exams","date":"2023-12-28","arxiv_id":"2312.16845","n_code_links":0,"syntology":null},{"paper":"/paper/principled-instructions-are-all-you-need-for","slug":"principled-instructions-are-all-you-need-for","title":"Principled Instructions Are All You Need for Questioning LLaMA-1/2, GPT-3.5/4","date":"2023-12-26","arxiv_id":"2312.16171","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vila-lab/atlas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/secqa-a-concise-question-answering-dataset","slug":"secqa-a-concise-question-answering-dataset","title":"SecQA: A Concise Question-Answering Dataset for Evaluating Large Language Models in Computer Security","date":"2023-12-26","arxiv_id":"2312.15838","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zefang-liu/lm-evaluation-harness"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"task-contamination-language-models-may-not-be","title":"Task Contamination: Language Models May Not Be Few-Shot Anymore","date":"2023-12-26","arxiv_id":"2312.16337","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficacy-of-machine-generated-instructions","title":"Efficacy of Machine-Generated Instructions","date":"2023-12-22","arxiv_id":"2312.14423","n_code_links":0,"syntology":null},{"paper":null,"slug":"fm-ov3d-foundation-model-based-cross-modal","title":"FM-OV3D: Foundation Model-based Cross-modal Knowledge Blending for Open-Vocabulary 3D Detection","date":"2023-12-22","arxiv_id":"2312.14465","n_code_links":0,"syntology":null},{"paper":"/paper/refining-gpt-3-embeddings-with-a-siamese","slug":"refining-gpt-3-embeddings-with-a-siamese","title":"Refining GPT-3 Embeddings with a Siamese Structure for Technical Post Duplicate Detection","date":"2023-12-22","arxiv_id":"2312.15068","n_code_links":1,"syntology":null},{"paper":"/paper/argue-with-me-tersely-towards-sentence-level","slug":"argue-with-me-tersely-towards-sentence-level","title":"Argue with Me Tersely: Towards Sentence-Level Counter-Argument Generation","date":"2023-12-21","arxiv_id":"2312.13608","n_code_links":1,"syntology":null},{"paper":"/paper/chatgpt-as-a-commenter-to-the-news-can-llms","slug":"chatgpt-as-a-commenter-to-the-news-can-llms","title":"ChatGPT as a commenter to the news: can LLMs generate human-like opinions?","date":"2023-12-21","arxiv_id":"2312.13961","n_code_links":1,"syntology":null},{"paper":null,"slug":"infovisdial-an-informative-visual-dialogue","title":"InfoVisDial: An Informative Visual Dialogue Dataset by Bridging Large Multimodal and Language Models","date":"2023-12-21","arxiv_id":"2312.13503","n_code_links":0,"syntology":null},{"paper":null,"slug":"team-irisapu-project-description-for-drc2023","title":"Team Irisapu Project Description for DRC2023","date":"2023-12-21","arxiv_id":"2312.13765","n_code_links":0,"syntology":null},{"paper":null,"slug":"typhoon-thai-large-language-models","title":"Typhoon: Thai Large Language Models","date":"2023-12-21","arxiv_id":"2312.13951","n_code_links":0,"syntology":null},{"paper":"/paper/agentcoder-multi-agent-based-code-generation","slug":"agentcoder-multi-agent-based-code-generation","title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","date":"2023-12-20","arxiv_id":"2312.13010","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["huangd1999/AgentCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"benchmarking-and-analyzing-in-context","title":"Benchmarking and Analyzing In-context Learning, Fine-tuning and Supervised Learning for Biomedical Knowledge Curation: a focused study on chemical entities of biological interest","date":"2023-12-20","arxiv_id":"2312.12989","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-chatgpt-be-your-personal-medical","title":"Can ChatGPT be Your Personal Medical Assistant?","date":"2023-12-19","arxiv_id":"2312.12006","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-in-medical-term","title":"Large Language Models in Medical Term Classification and Unexpected Misalignment Between Response and Reasoning","date":"2023-12-19","arxiv_id":"2312.14184","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-ai-vocational-skills-through","title":"Evaluating AI Vocational Skills Through Professional Testing","date":"2023-12-17","arxiv_id":"2312.10603","n_code_links":0,"syntology":null},{"paper":"/paper/hyperpie-hyperparameter-information","slug":"hyperpie-hyperparameter-information","title":"HyperPIE: Hyperparameter Information Extraction from Scientific Publications","date":"2023-12-17","arxiv_id":"2312.10638","n_code_links":1,"syntology":null},{"paper":null,"slug":"mixed-distillation-helps-smaller-language","title":"Mixed Distillation Helps Smaller Language Model Better Reasoning","date":"2023-12-17","arxiv_id":"2312.10730","n_code_links":0,"syntology":null},{"paper":"/paper/multi-label-classification-of-covid-tweets","slug":"multi-label-classification-of-covid-tweets","title":"Multi-Label Classification of COVID-Tweets Using Large Language Models","date":"2023-12-17","arxiv_id":"2312.10748","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comparative-analysis-of-large-language","title":"A Comparative Analysis of Large Language Models for Code Documentation Generation","date":"2023-12-16","arxiv_id":"2312.10349","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-novel-dataset-for-financial-education-text","title":"A Novel Dataset for Financial Education Text Simplification in Spanish","date":"2023-12-15","arxiv_id":"2312.09897","n_code_links":0,"syntology":null},{"paper":null,"slug":"distilling-large-language-models-for-matching","title":"Distilling Large Language Models for Matching Patients to Clinical Trials","date":"2023-12-15","arxiv_id":"2312.09958","n_code_links":0,"syntology":null},{"paper":"/paper/boosting-llm-reasoning-push-the-limits-of-few","slug":"boosting-llm-reasoning-push-the-limits-of-few","title":"Fewer is More: Boosting LLM Reasoning with Reinforced Context Pruning","date":"2023-12-14","arxiv_id":"2312.08901","n_code_links":0,"syntology":null},{"paper":null,"slug":"entity-augmented-code-generation","title":"Dynamic Retrieval-Augmented Generation","date":"2023-12-14","arxiv_id":"2312.08976","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-evaluation-improves-selective-generation","title":"Self-Evaluation Improves Selective Generation in Large Language Models","date":"2023-12-14","arxiv_id":"2312.09300","n_code_links":0,"syntology":null},{"paper":"/paper/tinygsm-achieving-80-on-gsm8k-with-small","slug":"tinygsm-achieving-80-on-gsm8k-with-small","title":"TinyGSM: achieving >80% on GSM8k with small language models","date":"2023-12-14","arxiv_id":"2312.09241","n_code_links":0,"syntology":null},{"paper":null,"slug":"weak-to-strong-generalization-eliciting","title":"Weak-to-Strong Generalization: Eliciting Strong Capabilities With Weak Supervision","date":"2023-12-14","arxiv_id":"2312.09390","n_code_links":0,"syntology":null},{"paper":"/paper/weaving-pathways-for-justice-with-gpt-llm","slug":"weaving-pathways-for-justice-with-gpt-llm","title":"Weaving Pathways for Justice with GPT: LLM-driven automated drafting of interactive legal applications","date":"2023-12-14","arxiv_id":"2312.09198","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-are-complex-table","title":"Large Language Models are Complex Table Parsers","date":"2023-12-13","arxiv_id":"2312.11521","n_code_links":0,"syntology":null},{"paper":"/paper/ai-control-improving-safety-despite","slug":"ai-control-improving-safety-despite","title":"AI Control: Improving Safety Despite Intentional Subversion","date":"2023-12-12","arxiv_id":"2312.06942","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rgreenblatt/control-evaluations"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/image-content-generation-with-causal","slug":"image-content-generation-with-causal","title":"Image Content Generation with Causal Reasoning","date":"2023-12-12","arxiv_id":"2312.07132","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":2,"n_instrument":2,"unverified":4,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ieit-agi/mix-shannon"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/multilingual-large-language-models-leak-human","slug":"multilingual-large-language-models-leak-human","title":"Multilingual large language models leak human stereotypes across language boundaries","date":"2023-12-12","arxiv_id":"2312.07141","n_code_links":1,"syntology":null},{"paper":"/paper/perseus-removing-energy-bloat-from-large","slug":"perseus-removing-energy-bloat-from-large","title":"Reducing Energy Bloat in Large Model Training","date":"2023-12-12","arxiv_id":"2312.06902","n_code_links":2,"syntology":null},{"paper":"/paper/can-it-edit-evaluating-the-ability-of-large","slug":"can-it-edit-evaluating-the-ability-of-large","title":"Can It Edit? Evaluating the Ability of Large Language Models to Follow Code Editing Instructions","date":"2023-12-11","arxiv_id":"2312.12450","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["nuprl/canitedit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-chatgpt-as-a-question-answering","title":"Evaluating ChatGPT as a Question Answering System: A Comprehensive Analysis and Comparison with Existing Models","date":"2023-12-11","arxiv_id":"2312.07592","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-large-language-models-are-all","title":"Generative Large Language Models Are All-purpose Text Analytics Engines: Text-to-text Learning Is All Your Need","date":"2023-12-11","arxiv_id":"2312.06099","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-limits-of-chatgpt-in-software","title":"Exploring the Limits of ChatGPT in Software Security Applications","date":"2023-12-08","arxiv_id":"2312.05275","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-interactive-optimization-of-open-source","title":"LLM Interactive Optimization of Open Source Python Libraries -- Case Studies and Generalization","date":"2023-12-08","arxiv_id":"2312.14949","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-sarcasm-detection-with-openai-gpt-based","title":"On Sarcasm Detection with OpenAI GPT-based Models","date":"2023-12-07","arxiv_id":"2312.04642","n_code_links":0,"syntology":null},{"paper":null,"slug":"holmes-towards-distributed-training-across","title":"Holmes: Towards Distributed Training Across Clusters with Heterogeneous NIC Environment","date":"2023-12-06","arxiv_id":"2312.03549","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-hardware-evaluation-framework-for-large","title":"A Hardware Evaluation Framework for Large Language Model Inference","date":"2023-12-05","arxiv_id":"2312.03134","n_code_links":0,"syntology":null},{"paper":"/paper/draft-dense-retrieval-augmented-few-shot","slug":"draft-dense-retrieval-augmented-few-shot","title":"DRAFT: Dense Retrieval Augmented Few-shot Topic classifier Framework","date":"2023-12-05","arxiv_id":"2312.02532","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-vs-human-for-scientific-reviews-a-dual","title":"GPT vs Human for Scientific Reviews: A Dual Source Review on Applications of ChatGPT in Science","date":"2023-12-05","arxiv_id":"2312.03769","n_code_links":0,"syntology":null},{"paper":null,"slug":"rank-without-gpt-building-gpt-independent","title":"Rank-without-GPT: Building GPT-Independent Listwise Rerankers on Open-Source Large Language Models","date":"2023-12-05","arxiv_id":"2312.02969","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-survey-on-large-language-model-llm-security","title":"A Survey on Large Language Model (LLM) Security and Privacy: The Good, the Bad, and the Ugly","date":"2023-12-04","arxiv_id":"2312.02003","n_code_links":0,"syntology":null},{"paper":null,"slug":"jellyfish-a-large-language-model-for-data","title":"Jellyfish: A Large Language Model for Data Preprocessing","date":"2023-12-04","arxiv_id":"2312.01678","n_code_links":0,"syntology":null},{"paper":"/paper/tree-of-attacks-jailbreaking-black-box-llms","slug":"tree-of-attacks-jailbreaking-black-box-llms","title":"Tree of Attacks: Jailbreaking Black-Box LLMs Automatically","date":"2023-12-04","arxiv_id":"2312.02119","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ricommunity/tap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/nlebench-norglm-a-comprehensive-empirical","slug":"nlebench-norglm-a-comprehensive-empirical","title":"NLEBench+NorGLM: A Comprehensive Empirical Analysis and Benchmark Dataset for Generative Language Models in Norwegian","date":"2023-12-03","arxiv_id":"2312.01314","n_code_links":1,"syntology":{"ran":0,"of":4,"n_ran_checked":0,"n_instrument":0,"unverified":4,"pointer_only":4,"phrase":"0 ran · 4 unverified","official":{"repos":["smartmedia-ai/norglm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"paper":"/paper/harnessing-the-power-of-prompt-based","slug":"harnessing-the-power-of-prompt-based","title":"Harnessing the Power of Prompt-based Techniques for Generating School-Level Questions using Large Language Models","date":"2023-12-02","arxiv_id":"2312.01032","n_code_links":1,"syntology":null},{"paper":null,"slug":"applying-large-language-models-and-chain-of","title":"Applying Large Language Models and Chain-of-Thought for Automatic Scoring","date":"2023-11-30","arxiv_id":"2312.03748","n_code_links":0,"syntology":null},{"paper":null,"slug":"iag-induction-augmented-generation-framework","title":"IAG: Induction-Augmented Generation Framework for Answering Reasoning Questions","date":"2023-11-30","arxiv_id":"2311.18397","n_code_links":0,"syntology":null},{"paper":"/paper/robust-concept-erasure-via-kernelized-rate-1","slug":"robust-concept-erasure-via-kernelized-rate-1","title":"Robust Concept Erasure via Kernelized Rate-Distortion Maximization","date":"2023-11-30","arxiv_id":"2312.00194","n_code_links":1,"syntology":{"ran":20,"of":21,"n_ran_checked":18,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["brcsomnath/kram"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/biomedical-knowledge-graph-enhanced-prompt","slug":"biomedical-knowledge-graph-enhanced-prompt","title":"Biomedical knowledge graph-optimized prompt generation for large language models","date":"2023-11-29","arxiv_id":"2311.17330","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":{"repos":["BaranziniLab/KG_RAG"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":"/paper/bert-goes-off-topic-investigating-the-domain","slug":"bert-goes-off-topic-investigating-the-domain","title":"BERT Goes Off-Topic: Investigating the Domain Transfer Challenge using Genre Classification","date":"2023-11-27","arxiv_id":"2311.16083","n_code_links":1,"syntology":null},{"paper":null,"slug":"decoding-logic-errors-a-comparative-study-on","title":"Decoding Logic Errors: A Comparative Study on Bug Detection by Students and Large Language Models","date":"2023-11-27","arxiv_id":"2311.16017","n_code_links":0,"syntology":null},{"paper":"/paper/machine-generated-text-detection-using-deep","slug":"machine-generated-text-detection-using-deep","title":"Machine-Generated Text Detection using Deep Learning","date":"2023-11-26","arxiv_id":"2311.15425","n_code_links":1,"syntology":null},{"paper":"/paper/gpt-struct-me-probing-gpt-models-on-narrative","slug":"gpt-struct-me-probing-gpt-models-on-narrative","title":"GPT Struct Me: Probing GPT Models on Narrative Entity Extraction","date":"2023-11-24","arxiv_id":"2311.14583","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-as-automated-aligners","title":"Large Language Models as Automated Aligners for benchmarking Vision-Language Models","date":"2023-11-24","arxiv_id":"2311.14580","n_code_links":0,"syntology":null},{"paper":null,"slug":"machine-translation-for-ge-ez-language","title":"Machine Translation for Ge'ez Language","date":"2023-11-24","arxiv_id":"2311.14530","n_code_links":0,"syntology":null},{"paper":"/paper/a-cross-attention-approach-to-diagnostic","slug":"a-cross-attention-approach-to-diagnostic","title":"A Cross Attention Approach to Diagnostic Explainability using Clinical Practice Guidelines for Depression","date":"2023-11-23","arxiv_id":"2311.13852","n_code_links":1,"syntology":null},{"paper":"/paper/hardware-resilience-properties-of-text-guided-1","slug":"hardware-resilience-properties-of-text-guided-1","title":"Hardware Resilience Properties of Text-Guided Image Classifiers","date":"2023-11-23","arxiv_id":"2311.14062","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["talalwasim/textguidedresilience"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"minimizing-factual-inconsistency-and","title":"Minimizing Factual Inconsistency and Hallucination in Large Language Models","date":"2023-11-23","arxiv_id":"2311.13878","n_code_links":0,"syntology":null},{"paper":null,"slug":"generation-of-explanations-for-logic","title":"Generation of Explanations for Logic Reasoning","date":"2023-11-22","arxiv_id":"2311.13455","n_code_links":0,"syntology":null},{"paper":null,"slug":"nova-generative-language-models-for-binaries","title":"Nova: Generative Language Models for Assembly Code with Hierarchical Attention and Contrastive Learning","date":"2023-11-22","arxiv_id":"2311.13721","n_code_links":0,"syntology":null},{"paper":"/paper/pg-video-llava-pixel-grounding-large-video","slug":"pg-video-llava-pixel-grounding-large-video","title":"PG-Video-LLaVA: Pixel Grounding Large Video-Language Models","date":"2023-11-22","arxiv_id":"2311.13435","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":2,"n_instrument":1,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mbzuai-oryx/video-llava"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/speak-like-a-native-prompting-large-language","slug":"speak-like-a-native-prompting-large-language","title":"AlignedCoT: Prompting Large Language Models via Native-Speaking Demonstrations","date":"2023-11-22","arxiv_id":"2311.13538","n_code_links":1,"syntology":null},{"paper":null,"slug":"ve-a-chatbot-for-latin","title":"@ve: A Chatbot for Latin","date":"2023-11-22","arxiv_id":"2311.14741","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-survey-on-large-language-models-for-1","title":"A Survey on Large Language Models for Personalized and Explainable Recommendations","date":"2023-11-21","arxiv_id":"2311.12338","n_code_links":0,"syntology":null},{"paper":null,"slug":"interprompt-interpretable-prompting-for","title":"InterPrompt: Interpretable Prompting for Interrelated Interpersonal Risk Factors in Reddit Posts","date":"2023-11-21","arxiv_id":"2311.12404","n_code_links":0,"syntology":null}],"record_sha256":"748e2d9f560b2f658c5bf23079ae4ba7d5231925701579bb06ea354fb55fda8b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}