{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/cosine-annealing/papers/18","list_of":"/method/cosine-annealing","method":"Cosine Annealing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":18,"pages_in_order":40,"rows_per_page":100,"rows":[1701,1800],"of":3965,"counts":{"archive_papers_tagged":3965,"with_a_code_link":1734,"where_syntology_ran_a_sample":627,"not_listed_spam_title":0,"listed":3965,"listed_where_code_ran":627,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":513,"every_run_a_failure_of_syntologys_instrument":114,"listed_with_a_run_with_no_instrument_failure":513,"listed_every_run_a_failure_of_syntologys_instrument":114,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/cosine-annealing","prev":"/method/cosine-annealing/papers/17","next":"/method/cosine-annealing/papers/19","papers":[{"paper":"/paper/exploiting-inter-layer-expert-affinity-for","slug":"exploiting-inter-layer-expert-affinity-for","title":"Exploiting Inter-Layer Expert Affinity for Accelerating Mixture-of-Experts Model Inference","date":"2024-01-16","arxiv_id":"2401.08383","n_code_links":1,"syntology":null},{"paper":null,"slug":"rag-vs-fine-tuning-pipelines-tradeoffs-and-a","title":"RAG vs Fine-tuning: Pipelines, Tradeoffs, and a Case Study on Agriculture","date":"2024-01-16","arxiv_id":"2401.08406","n_code_links":0,"syntology":null},{"paper":"/paper/rotbench-a-multi-level-benchmark-for","slug":"rotbench-a-multi-level-benchmark-for","title":"RoTBench: A Multi-Level Benchmark for Evaluating the Robustness of Large Language Models in Tool Learning","date":"2024-01-16","arxiv_id":"2401.08326","n_code_links":1,"syntology":{"ran":10,"of":15,"n_ran_checked":10,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["junjie-ye/rotbench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/tuning-language-models-by-proxy","slug":"tuning-language-models-by-proxy","title":"Tuning Language Models by Proxy","date":"2024-01-16","arxiv_id":"2401.08565","n_code_links":2,"syntology":{"ran":9,"of":10,"n_ran_checked":7,"n_instrument":2,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["alisawuffles/proxy-tuning"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-novel-approach-for-automatic-program-repair","slug":"a-novel-approach-for-automatic-program-repair","title":"A Novel Approach for Automatic Program Repair using Round-Trip Translation with Large Language Models","date":"2024-01-15","arxiv_id":"2401.07994","n_code_links":1,"syntology":null},{"paper":null,"slug":"survival-analysis-of-young-triple-negative","title":"Survival Analysis of Young Triple-Negative Breast Cancer Patients","date":"2024-01-15","arxiv_id":"2401.08712","n_code_links":0,"syntology":null},{"paper":"/paper/harnessing-large-language-models-over","slug":"harnessing-large-language-models-over","title":"Harnessing Large Language Models Over Transformer Models for Detecting Bengali Depressive Social Media Text: A Comprehensive Study","date":"2024-01-14","arxiv_id":"2401.07310","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-be-homo-economicus-can-an-llm","title":"Learning to be Homo Economicus: Can an LLM Learn Preferences from Choice","date":"2024-01-14","arxiv_id":"2401.07345","n_code_links":0,"syntology":null},{"paper":null,"slug":"mapgpt-map-guided-prompting-for-unified","title":"MapGPT: Map-Guided Prompting with Adaptive Path Planning for Vision-and-Language Navigation","date":"2024-01-14","arxiv_id":"2401.07314","n_code_links":0,"syntology":null},{"paper":null,"slug":"streamlining-the-selection-phase-of","title":"Streamlining the Selection Phase of Systematic Literature Reviews (SLRs) Using AI-Enabled GPT-4 Assistant API","date":"2024-01-14","arxiv_id":"2402.18582","n_code_links":0,"syntology":null},{"paper":"/paper/a-novel-multi-stage-prompting-approach-for","slug":"a-novel-multi-stage-prompting-approach-for","title":"A Novel Multi-Stage Prompting Approach for Language Agnostic MCQ Generation using GPT","date":"2024-01-13","arxiv_id":"2401.07098","n_code_links":1,"syntology":null},{"paper":null,"slug":"assessing-large-language-models-in-mechanical","title":"Assessing Large Language Models in Mechanical Engineering Education: A Study on Mechanics-Focused Conceptual Understanding","date":"2024-01-13","arxiv_id":"2401.12983","n_code_links":0,"syntology":null},{"paper":null,"slug":"combining-confidence-elicitation-and-sample","title":"Combining Confidence Elicitation and Sample-based Methods for Uncertainty Quantification in Misinformation Mitigation","date":"2024-01-13","arxiv_id":"2401.08694","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-gpt-4-and-open-source-language","title":"Comparing GPT-4 and Open-Source Language Models in Misinformation Mitigation","date":"2024-01-12","arxiv_id":"2401.06920","n_code_links":0,"syntology":null},{"paper":"/paper/from-automation-to-augmentation-large","slug":"from-automation-to-augmentation-large","title":"Human-AI Collaborative Essay Scoring: A Dual-Process Framework with LLMs","date":"2024-01-12","arxiv_id":"2401.06431","n_code_links":1,"syntology":null},{"paper":"/paper/how-johnny-can-persuade-llms-to-jailbreak","slug":"how-johnny-can-persuade-llms-to-jailbreak","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","date":"2024-01-12","arxiv_id":"2401.06373","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chats-lab/persuasive_jailbreaker"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/intention-analysis-prompting-makes-large","slug":"intention-analysis-prompting-makes-large","title":"Intention Analysis Makes LLMs A Good Jailbreak Defender","date":"2024-01-12","arxiv_id":"2401.06561","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["alphadl/safellm_with_intentionanalysis"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/mission-impossible-language-models","slug":"mission-impossible-language-models","title":"Mission: Impossible Language Models","date":"2024-01-12","arxiv_id":"2401.06416","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":10,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jkallini/mission-impossible-language-models"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"persianmind-a-cross-lingual-persian-english","title":"PersianMind: A Cross-Lingual Persian-English Large Language Model","date":"2024-01-12","arxiv_id":"2401.06466","n_code_links":0,"syntology":null},{"paper":"/paper/pizzacommonsense-learning-to-model","slug":"pizzacommonsense-learning-to-model","title":"PizzaCommonSense: Learning to Model Commonsense Reasoning about Intermediate Steps in Cooking Recipes","date":"2024-01-12","arxiv_id":"2401.06930","n_code_links":1,"syntology":null},{"paper":null,"slug":"investigating-data-contamination-for-pre","title":"Investigating Data Contamination for Pre-training Language Models","date":"2024-01-11","arxiv_id":"2401.06059","n_code_links":0,"syntology":null},{"paper":null,"slug":"mutation-based-consistency-testing-for","title":"Mutation-based Consistency Testing for Evaluating the Code Understanding Capability of LLMs","date":"2024-01-11","arxiv_id":"2401.05940","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-based-mental-health-screening-from","title":"Prompt-based mental health screening from social media text","date":"2024-01-11","arxiv_id":"2401.05912","n_code_links":0,"syntology":null},{"paper":"/paper/the-benefits-of-a-concise-chain-of-thought-on","slug":"the-benefits-of-a-concise-chain-of-thought-on","title":"The Benefits of a Concise Chain of Thought on Problem-Solving in Large Language Models","date":"2024-01-11","arxiv_id":"2401.05618","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["matthewrenze/jhu-concise-cot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/yolo-former-yolo-shakes-hand-with-vit","slug":"yolo-former-yolo-shakes-hand-with-vit","title":"YOLO-Former: YOLO Shakes Hand With ViT","date":"2024-01-11","arxiv_id":"2401.06244","n_code_links":0,"syntology":null},{"paper":"/paper/autoact-automatic-agent-learning-from-scratch","slug":"autoact-automatic-agent-learning-from-scratch","title":"AutoAct: Automatic Agent Learning from Scratch for QA via Self-Planning","date":"2024-01-10","arxiv_id":"2401.05268","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":6,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zjunlp/autoact"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/i-am-a-strange-dataset-metalinguistic-tests","slug":"i-am-a-strange-dataset-metalinguistic-tests","title":"I am a Strange Dataset: Metalinguistic Tests for Language Models","date":"2024-01-10","arxiv_id":"2401.05300","n_code_links":1,"syntology":null},{"paper":"/paper/infiagent-dabench-evaluating-agents-on-data","slug":"infiagent-dabench-evaluating-agents-on-data","title":"InfiAgent-DABench: Evaluating Agents on Data Analysis Tasks","date":"2024-01-10","arxiv_id":"2401.05507","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["infiagent/infiagent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"machine-teaching-for-building-modular-ai","title":"Can Active Label Correction Improve LLM-based Modular AI Systems?","date":"2024-01-10","arxiv_id":"2401.05467","n_code_links":0,"syntology":null},{"paper":null,"slug":"monte-carlo-tree-search-for-recipe-generation","title":"Monte Carlo Tree Search for Recipe Generation using GPT-2","date":"2024-01-10","arxiv_id":"2401.05199","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-optimizing-rag-for","title":"Reinforcement Learning for Optimizing RAG for Domain Chatbots","date":"2024-01-10","arxiv_id":"2401.06800","n_code_links":0,"syntology":null},{"paper":null,"slug":"fighting-fire-with-fire-adversarial-prompting","title":"Fighting Fire with Fire: Adversarial Prompting to Generate a Misinformation Detection Dataset","date":"2024-01-09","arxiv_id":"2401.04481","n_code_links":0,"syntology":null},{"paper":"/paper/advancing-spatial-reasoning-in-large-language","slug":"advancing-spatial-reasoning-in-large-language","title":"Advancing Spatial Reasoning in Large Language Models: An In-Depth Evaluation and Enhancement Using the StepGame Benchmark","date":"2024-01-08","arxiv_id":"2401.03991","n_code_links":1,"syntology":null},{"paper":null,"slug":"distortions-in-judged-spatial-relations-in","title":"Distortions in Judged Spatial Relations in Large Language Models","date":"2024-01-08","arxiv_id":"2401.04218","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-in-bioinformatics","title":"Advancing bioinformatics with large language models: components, applications and perspectives","date":"2024-01-08","arxiv_id":"2401.04155","n_code_links":0,"syntology":null},{"paper":"/paper/llm4plc-harnessing-large-language-models-for","slug":"llm4plc-harnessing-large-language-models-for","title":"LLM4PLC: Harnessing Large Language Models for Verifiable Programming of PLCs in Industrial Control Systems","date":"2024-01-08","arxiv_id":"2401.05443","n_code_links":1,"syntology":null},{"paper":"/paper/mixtral-of-experts","slug":"mixtral-of-experts","title":"Mixtral of Experts","date":"2024-01-08","arxiv_id":"2401.04088","n_code_links":6,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","official":null}},{"paper":null,"slug":"d-causal-exploring-defeasibility-in-causal","title":"Exploring Defeasibility in Causal Reasoning","date":"2024-01-06","arxiv_id":"2401.03183","n_code_links":0,"syntology":null},{"paper":null,"slug":"pixar-auto-regressive-language-modeling-in","title":"PIXAR: Auto-Regressive Language Modeling in Pixel Space","date":"2024-01-06","arxiv_id":"2401.03321","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-large-language-models-to-assess-tutors","title":"Using Large Language Models to Assess Tutors' Performance in Reacting to Students Making Math Errors","date":"2024-01-06","arxiv_id":"2401.03238","n_code_links":0,"syntology":null},{"paper":"/paper/comparative-analysis-of-llama-and-chatgpt","slug":"comparative-analysis-of-llama-and-chatgpt","title":"Can Large Language Models Understand Molecules?","date":"2024-01-05","arxiv_id":"2402.00024","n_code_links":2,"syntology":null},{"paper":"/paper/deepseek-llm-scaling-open-source-language","slug":"deepseek-llm-scaling-open-source-language","title":"DeepSeek LLM: Scaling Open-Source Language Models with Longtermism","date":"2024-01-05","arxiv_id":"2401.02954","n_code_links":1,"syntology":null},{"paper":"/paper/parameter-efficient-sparsity-crafting-from","slug":"parameter-efficient-sparsity-crafting-from","title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","date":"2024-01-05","arxiv_id":"2401.02731","n_code_links":2,"syntology":null},{"paper":null,"slug":"are-llms-robust-for-spoken-dialogues","title":"Are LLMs Robust for Spoken Dialogues?","date":"2024-01-04","arxiv_id":"2401.02297","n_code_links":0,"syntology":null},{"paper":null,"slug":"hypersense-accelerating-hyper-dimensional","title":"HyperSense: Hyperdimensional Intelligent Sensing for Energy-Efficient Sparse Data Processing","date":"2024-01-04","arxiv_id":"2401.10267","n_code_links":0,"syntology":null},{"paper":null,"slug":"re-evaluating-the-memory-balanced-pipeline","title":"Re-evaluating the Memory-balanced Pipeline Parallelism: BPipe","date":"2024-01-04","arxiv_id":"2401.02088","n_code_links":0,"syntology":null},{"paper":"/paper/text2mdt-extracting-medical-decision-trees","slug":"text2mdt-extracting-medical-decision-trees","title":"Text2MDT: Extracting Medical Decision Trees from Medical Texts","date":"2024-01-04","arxiv_id":"2401.02034","n_code_links":1,"syntology":null},{"paper":null,"slug":"iot-in-the-era-of-generative-ai-vision-and","title":"The Internet of Things in the Era of Generative AI: Vision and Challenges","date":"2024-01-03","arxiv_id":"2401.01923","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-zero-shot-abstractive","slug":"revisiting-zero-shot-abstractive","title":"Revisiting Zero-Shot Abstractive Summarization in the Era of Large Language Models from the Perspective of Position Bias","date":"2024-01-03","arxiv_id":"2401.01989","n_code_links":1,"syntology":null},{"paper":"/paper/vietnamese-poem-generation-the-prospect-of","slug":"vietnamese-poem-generation-the-prospect-of","title":"Vietnamese Poem Generation & The Prospect Of Cross-Language Poem-To-Poem Translation","date":"2024-01-02","arxiv_id":"2401.01078","n_code_links":1,"syntology":null},{"paper":"/paper/a-b-b-a-triggering-logical-reasoning-failures","slug":"a-b-b-a-triggering-logical-reasoning-failures","title":"LogicAsker: Evaluating and Improving the Logical Reasoning Ability of Large Language Models","date":"2024-01-01","arxiv_id":"2401.00757","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yxwan123/logicasker"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-computational-framework-for-behavioral","slug":"a-computational-framework-for-behavioral","title":"A Computational Framework for Behavioral Assessment of LLM Therapists","date":"2024-01-01","arxiv_id":"2401.00820","n_code_links":1,"syntology":null},{"paper":"/paper/adapt-or-perish-adaptive-sparse-transformer","slug":"adapt-or-perish-adaptive-sparse-transformer","title":"Adapt or Perish: Adaptive Sparse Transformer with Attentive Feature Refinement for Image Restoration","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-aren-t-all-that-you","title":"Large Language Models aren't all that you need","date":"2024-01-01","arxiv_id":"2401.00698","n_code_links":0,"syntology":null},{"paper":"/paper/seed-bench-benchmarking-multimodal-large","slug":"seed-bench-benchmarking-multimodal-large","title":"SEED-Bench: Benchmarking Multimodal Large Language Models","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/advancing-ttp-analysis-harnessing-the-power","slug":"advancing-ttp-analysis-harnessing-the-power","title":"Advancing TTP Analysis: Harnessing the Power of Large Language Models with Retrieval Augmented Generation","date":"2023-12-30","arxiv_id":"2401.00280","n_code_links":1,"syntology":null},{"paper":null,"slug":"trace-and-edit-relation-associations-in-gpt","title":"Trace and Edit Relation Associations in GPT","date":"2023-12-30","arxiv_id":"2401.02976","n_code_links":0,"syntology":null},{"paper":"/paper/gemini-in-reasoning-unveiling-commonsense-in","slug":"gemini-in-reasoning-unveiling-commonsense-in","title":"Gemini in Reasoning: Unveiling Commonsense in Multimodal Large Language Models","date":"2023-12-29","arxiv_id":"2312.17661","n_code_links":1,"syntology":null},{"paper":"/paper/jatmo-prompt-injection-defense-by-task","slug":"jatmo-prompt-injection-defense-by-task","title":"Jatmo: Prompt Injection Defense by Task-Specific Finetuning","date":"2023-12-29","arxiv_id":"2312.17673","n_code_links":1,"syntology":{"ran":10,"of":18,"n_ran_checked":10,"n_instrument":0,"unverified":8,"pointer_only":18,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","official":{"repos":["wagner-group/prompt-injection-defense"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-the-performance-of-large-language-1","title":"Evaluating the Performance of Large Language Models for Spanish Language in Undergraduate Admissions Exams","date":"2023-12-28","arxiv_id":"2312.16845","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-model-for-causal-decision","title":"LLM4Causal: Democratized Causal Tools for Everyone via Large Language Model","date":"2023-12-28","arxiv_id":"2312.17122","n_code_links":0,"syntology":null},{"paper":null,"slug":"pangu-p-enhancing-language-model","title":"PanGu-$π$: Enhancing Language Model Architectures via Nonlinearity Compensation","date":"2023-12-27","arxiv_id":"2312.17276","n_code_links":0,"syntology":null},{"paper":null,"slug":"chartbench-a-benchmark-for-complex-visual","title":"ChartBench: A Benchmark for Complex Visual Reasoning in Charts","date":"2023-12-26","arxiv_id":"2312.15915","n_code_links":0,"syntology":null},{"paper":"/paper/principled-instructions-are-all-you-need-for","slug":"principled-instructions-are-all-you-need-for","title":"Principled Instructions Are All You Need for Questioning LLaMA-1/2, GPT-3.5/4","date":"2023-12-26","arxiv_id":"2312.16171","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vila-lab/atlas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/secqa-a-concise-question-answering-dataset","slug":"secqa-a-concise-question-answering-dataset","title":"SecQA: A Concise Question-Answering Dataset for Evaluating Large Language Models in Computer Security","date":"2023-12-26","arxiv_id":"2312.15838","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zefang-liu/lm-evaluation-harness"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"task-contamination-language-models-may-not-be","title":"Task Contamination: Language Models May Not Be Few-Shot Anymore","date":"2023-12-26","arxiv_id":"2312.16337","n_code_links":0,"syntology":null},{"paper":"/paper/fairness-aware-structured-pruning-in","slug":"fairness-aware-structured-pruning-in","title":"Fairness-Aware Structured Pruning in Transformers","date":"2023-12-24","arxiv_id":"2312.15398","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["chandar-lab/fasp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"do-llm-agents-exhibit-social-behavior","title":"Do LLM Agents Exhibit Social Behavior?","date":"2023-12-23","arxiv_id":"2312.15198","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-the-potential-of-fpga-based","slug":"understanding-the-potential-of-fpga-based","title":"Understanding the Potential of FPGA-Based Spatial Acceleration for Large Language Model Inference","date":"2023-12-23","arxiv_id":"2312.15159","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficacy-of-machine-generated-instructions","title":"Efficacy of Machine-Generated Instructions","date":"2023-12-22","arxiv_id":"2312.14423","n_code_links":0,"syntology":null},{"paper":null,"slug":"fm-ov3d-foundation-model-based-cross-modal","title":"FM-OV3D: Foundation Model-based Cross-modal Knowledge Blending for Open-Vocabulary 3D Detection","date":"2023-12-22","arxiv_id":"2312.14465","n_code_links":0,"syntology":null},{"paper":"/paper/refining-gpt-3-embeddings-with-a-siamese","slug":"refining-gpt-3-embeddings-with-a-siamese","title":"Refining GPT-3 Embeddings with a Siamese Structure for Technical Post Duplicate Detection","date":"2023-12-22","arxiv_id":"2312.15068","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-the-regularity-of-self","title":"How Smooth Is Attention?","date":"2023-12-22","arxiv_id":"2312.14820","n_code_links":0,"syntology":null},{"paper":"/paper/argue-with-me-tersely-towards-sentence-level","slug":"argue-with-me-tersely-towards-sentence-level","title":"Argue with Me Tersely: Towards Sentence-Level Counter-Argument Generation","date":"2023-12-21","arxiv_id":"2312.13608","n_code_links":1,"syntology":null},{"paper":"/paper/chatgpt-as-a-commenter-to-the-news-can-llms","slug":"chatgpt-as-a-commenter-to-the-news-can-llms","title":"ChatGPT as a commenter to the news: can LLMs generate human-like opinions?","date":"2023-12-21","arxiv_id":"2312.13961","n_code_links":1,"syntology":null},{"paper":"/paper/de-novo-drug-design-using-reinforcement-1","slug":"de-novo-drug-design-using-reinforcement-1","title":"De novo Drug Design using Reinforcement Learning with Multiple GPT Agents","date":"2023-12-21","arxiv_id":"2401.06155","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hxyfighter/molrl-mgpt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"infovisdial-an-informative-visual-dialogue","title":"InfoVisDial: An Informative Visual Dialogue Dataset by Bridging Large Multimodal and Language Models","date":"2023-12-21","arxiv_id":"2312.13503","n_code_links":0,"syntology":null},{"paper":"/paper/provfl-client-driven-interpretability-of","slug":"provfl-client-driven-interpretability-of","title":"TraceFL: Interpretability-Driven Debugging in Federated Learning via Neuron Provenance","date":"2023-12-21","arxiv_id":"2312.13632","n_code_links":2,"syntology":null},{"paper":null,"slug":"team-irisapu-project-description-for-drc2023","title":"Team Irisapu Project Description for DRC2023","date":"2023-12-21","arxiv_id":"2312.13765","n_code_links":0,"syntology":null},{"paper":null,"slug":"typhoon-thai-large-language-models","title":"Typhoon: Thai Large Language Models","date":"2023-12-21","arxiv_id":"2312.13951","n_code_links":0,"syntology":null},{"paper":"/paper/agentcoder-multi-agent-based-code-generation","slug":"agentcoder-multi-agent-based-code-generation","title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","date":"2023-12-20","arxiv_id":"2312.13010","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["huangd1999/AgentCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"benchmarking-and-analyzing-in-context","title":"Benchmarking and Analyzing In-context Learning, Fine-tuning and Supervised Learning for Biomedical Knowledge Curation: a focused study on chemical entities of biological interest","date":"2023-12-20","arxiv_id":"2312.12989","n_code_links":0,"syntology":null},{"paper":"/paper/domain-specific-code-language-models","slug":"domain-specific-code-language-models","title":"MonoCoder: Domain-Specific Code Language Model for HPC Codes and Tasks","date":"2023-12-20","arxiv_id":"2312.13322","n_code_links":3,"syntology":null},{"paper":null,"slug":"can-chatgpt-be-your-personal-medical","title":"Can ChatGPT be Your Personal Medical Assistant?","date":"2023-12-19","arxiv_id":"2312.12006","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-transformers-learn-sequential-function","title":"Can Transformers Learn Sequential Function Classes In Context?","date":"2023-12-19","arxiv_id":"2312.12655","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-in-medical-term","title":"Large Language Models in Medical Term Classification and Unexpected Misalignment Between Response and Reasoning","date":"2023-12-19","arxiv_id":"2312.14184","n_code_links":0,"syntology":null},{"paper":"/paper/an-in-depth-look-at-gemini-s-language","slug":"an-in-depth-look-at-gemini-s-language","title":"An In-depth Look at Gemini's Language Abilities","date":"2023-12-18","arxiv_id":"2312.11444","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":11,"n_instrument":0,"unverified":1,"pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["neulab/gemini-benchmark"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generative-linguistic-representation-for","title":"Generative linguistic representation for spoken language identification","date":"2023-12-18","arxiv_id":"2312.10964","n_code_links":0,"syntology":null},{"paper":null,"slug":"artificial-intelligence-optical-hardware","title":"Artificial intelligence optical hardware empowers high-resolution hyperspectral video understanding at 1.2 Tb/s","date":"2023-12-17","arxiv_id":"2312.10639","n_code_links":0,"syntology":null},{"paper":"/paper/decoding-concerns-multi-label-classification","slug":"decoding-concerns-multi-label-classification","title":"Decoding Concerns: Multi-label Classification of Vaccine Sentiments in Social Media","date":"2023-12-17","arxiv_id":"2312.10626","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-ai-vocational-skills-through","title":"Evaluating AI Vocational Skills Through Professional Testing","date":"2023-12-17","arxiv_id":"2312.10603","n_code_links":0,"syntology":null},{"paper":"/paper/hyperpie-hyperparameter-information","slug":"hyperpie-hyperparameter-information","title":"HyperPIE: Hyperparameter Information Extraction from Scientific Publications","date":"2023-12-17","arxiv_id":"2312.10638","n_code_links":1,"syntology":null},{"paper":null,"slug":"mixed-distillation-helps-smaller-language","title":"Mixed Distillation Helps Smaller Language Model Better Reasoning","date":"2023-12-17","arxiv_id":"2312.10730","n_code_links":0,"syntology":null},{"paper":"/paper/multi-label-classification-of-covid-tweets","slug":"multi-label-classification-of-covid-tweets","title":"Multi-Label Classification of COVID-Tweets Using Large Language Models","date":"2023-12-17","arxiv_id":"2312.10748","n_code_links":1,"syntology":null},{"paper":null,"slug":"t2m-hifigpt-generating-high-quality-human","title":"T2M-HiFiGPT: Generating High Quality Human Motion from Textual Descriptions with Residual Discrete Representations","date":"2023-12-17","arxiv_id":"2312.10628","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-analysis-of-large-language","title":"A Comparative Analysis of Large Language Models for Code Documentation Generation","date":"2023-12-16","arxiv_id":"2312.10349","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-novel-dataset-for-financial-education-text","title":"A Novel Dataset for Financial Education Text Simplification in Spanish","date":"2023-12-15","arxiv_id":"2312.09897","n_code_links":0,"syntology":null},{"paper":null,"slug":"distilling-large-language-models-for-matching","title":"Distilling Large Language Models for Matching Patients to Clinical Trials","date":"2023-12-15","arxiv_id":"2312.09958","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-multi-level-threats-in-telegram","slug":"exploring-multi-level-threats-in-telegram","title":"Exploring Multi-Level Threats in Telegram Data with AI-Human Annotation: A Preliminary Study","date":"2023-12-15","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"red-ai-inconsistent-responses-from-gpt3-5","title":"Red AI? Inconsistent Responses from GPT3.5 Models on Political Issues in the US and China","date":"2023-12-15","arxiv_id":"2312.09917","n_code_links":0,"syntology":null}],"record_sha256":"524cea4ba46a48f4bdc5fc8d1c8d369976c779c6c5d8c3c24b4357a5100a4159","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}