{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-cosine-annealing/papers/18","list_of":"/method/linear-warmup-with-cosine-annealing","method":"Linear Warmup With Cosine Annealing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":18,"pages_in_order":38,"rows_per_page":100,"rows":[1701,1800],"of":3797,"counts":{"archive_papers_tagged":3797,"with_a_code_link":1655,"where_syntology_ran_a_sample":602,"not_listed_spam_title":0,"listed":3797,"listed_where_code_ran":602,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":490,"every_run_a_failure_of_syntologys_instrument":112,"listed_with_a_run_with_no_instrument_failure":490,"listed_every_run_a_failure_of_syntologys_instrument":112,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-cosine-annealing","prev":"/method/linear-warmup-with-cosine-annealing/papers/17","next":"/method/linear-warmup-with-cosine-annealing/papers/19","papers":[{"paper":"/paper/i-am-a-strange-dataset-metalinguistic-tests","slug":"i-am-a-strange-dataset-metalinguistic-tests","title":"I am a Strange Dataset: Metalinguistic Tests for Language Models","date":"2024-01-10","arxiv_id":"2401.05300","n_code_links":1,"syntology":null},{"paper":"/paper/infiagent-dabench-evaluating-agents-on-data","slug":"infiagent-dabench-evaluating-agents-on-data","title":"InfiAgent-DABench: Evaluating Agents on Data Analysis Tasks","date":"2024-01-10","arxiv_id":"2401.05507","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["infiagent/infiagent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"machine-teaching-for-building-modular-ai","title":"Can Active Label Correction Improve LLM-based Modular AI Systems?","date":"2024-01-10","arxiv_id":"2401.05467","n_code_links":0,"syntology":null},{"paper":null,"slug":"monte-carlo-tree-search-for-recipe-generation","title":"Monte Carlo Tree Search for Recipe Generation using GPT-2","date":"2024-01-10","arxiv_id":"2401.05199","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-optimizing-rag-for","title":"Reinforcement Learning for Optimizing RAG for Domain Chatbots","date":"2024-01-10","arxiv_id":"2401.06800","n_code_links":0,"syntology":null},{"paper":null,"slug":"fighting-fire-with-fire-adversarial-prompting","title":"Fighting Fire with Fire: Adversarial Prompting to Generate a Misinformation Detection Dataset","date":"2024-01-09","arxiv_id":"2401.04481","n_code_links":0,"syntology":null},{"paper":"/paper/advancing-spatial-reasoning-in-large-language","slug":"advancing-spatial-reasoning-in-large-language","title":"Advancing Spatial Reasoning in Large Language Models: An In-Depth Evaluation and Enhancement Using the StepGame Benchmark","date":"2024-01-08","arxiv_id":"2401.03991","n_code_links":1,"syntology":null},{"paper":null,"slug":"distortions-in-judged-spatial-relations-in","title":"Distortions in Judged Spatial Relations in Large Language Models","date":"2024-01-08","arxiv_id":"2401.04218","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-in-bioinformatics","title":"Advancing bioinformatics with large language models: components, applications and perspectives","date":"2024-01-08","arxiv_id":"2401.04155","n_code_links":0,"syntology":null},{"paper":"/paper/llm4plc-harnessing-large-language-models-for","slug":"llm4plc-harnessing-large-language-models-for","title":"LLM4PLC: Harnessing Large Language Models for Verifiable Programming of PLCs in Industrial Control Systems","date":"2024-01-08","arxiv_id":"2401.05443","n_code_links":1,"syntology":null},{"paper":"/paper/mixtral-of-experts","slug":"mixtral-of-experts","title":"Mixtral of Experts","date":"2024-01-08","arxiv_id":"2401.04088","n_code_links":6,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","official":null}},{"paper":null,"slug":"d-causal-exploring-defeasibility-in-causal","title":"Exploring Defeasibility in Causal Reasoning","date":"2024-01-06","arxiv_id":"2401.03183","n_code_links":0,"syntology":null},{"paper":null,"slug":"pixar-auto-regressive-language-modeling-in","title":"PIXAR: Auto-Regressive Language Modeling in Pixel Space","date":"2024-01-06","arxiv_id":"2401.03321","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-large-language-models-to-assess-tutors","title":"Using Large Language Models to Assess Tutors' Performance in Reacting to Students Making Math Errors","date":"2024-01-06","arxiv_id":"2401.03238","n_code_links":0,"syntology":null},{"paper":"/paper/comparative-analysis-of-llama-and-chatgpt","slug":"comparative-analysis-of-llama-and-chatgpt","title":"Can Large Language Models Understand Molecules?","date":"2024-01-05","arxiv_id":"2402.00024","n_code_links":2,"syntology":null},{"paper":"/paper/deepseek-llm-scaling-open-source-language","slug":"deepseek-llm-scaling-open-source-language","title":"DeepSeek LLM: Scaling Open-Source Language Models with Longtermism","date":"2024-01-05","arxiv_id":"2401.02954","n_code_links":1,"syntology":null},{"paper":"/paper/parameter-efficient-sparsity-crafting-from","slug":"parameter-efficient-sparsity-crafting-from","title":"Parameter-Efficient Sparsity Crafting from Dense to Mixture-of-Experts for Instruction Tuning on General Tasks","date":"2024-01-05","arxiv_id":"2401.02731","n_code_links":2,"syntology":null},{"paper":null,"slug":"are-llms-robust-for-spoken-dialogues","title":"Are LLMs Robust for Spoken Dialogues?","date":"2024-01-04","arxiv_id":"2401.02297","n_code_links":0,"syntology":null},{"paper":null,"slug":"re-evaluating-the-memory-balanced-pipeline","title":"Re-evaluating the Memory-balanced Pipeline Parallelism: BPipe","date":"2024-01-04","arxiv_id":"2401.02088","n_code_links":0,"syntology":null},{"paper":"/paper/text2mdt-extracting-medical-decision-trees","slug":"text2mdt-extracting-medical-decision-trees","title":"Text2MDT: Extracting Medical Decision Trees from Medical Texts","date":"2024-01-04","arxiv_id":"2401.02034","n_code_links":1,"syntology":null},{"paper":null,"slug":"iot-in-the-era-of-generative-ai-vision-and","title":"The Internet of Things in the Era of Generative AI: Vision and Challenges","date":"2024-01-03","arxiv_id":"2401.01923","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-zero-shot-abstractive","slug":"revisiting-zero-shot-abstractive","title":"Revisiting Zero-Shot Abstractive Summarization in the Era of Large Language Models from the Perspective of Position Bias","date":"2024-01-03","arxiv_id":"2401.01989","n_code_links":1,"syntology":null},{"paper":"/paper/vietnamese-poem-generation-the-prospect-of","slug":"vietnamese-poem-generation-the-prospect-of","title":"Vietnamese Poem Generation & The Prospect Of Cross-Language Poem-To-Poem Translation","date":"2024-01-02","arxiv_id":"2401.01078","n_code_links":1,"syntology":null},{"paper":"/paper/a-b-b-a-triggering-logical-reasoning-failures","slug":"a-b-b-a-triggering-logical-reasoning-failures","title":"LogicAsker: Evaluating and Improving the Logical Reasoning Ability of Large Language Models","date":"2024-01-01","arxiv_id":"2401.00757","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yxwan123/logicasker"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-computational-framework-for-behavioral","slug":"a-computational-framework-for-behavioral","title":"A Computational Framework for Behavioral Assessment of LLM Therapists","date":"2024-01-01","arxiv_id":"2401.00820","n_code_links":1,"syntology":null},{"paper":"/paper/adapt-or-perish-adaptive-sparse-transformer","slug":"adapt-or-perish-adaptive-sparse-transformer","title":"Adapt or Perish: Adaptive Sparse Transformer with Attentive Feature Refinement for Image Restoration","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-aren-t-all-that-you","title":"Large Language Models aren't all that you need","date":"2024-01-01","arxiv_id":"2401.00698","n_code_links":0,"syntology":null},{"paper":"/paper/seed-bench-benchmarking-multimodal-large","slug":"seed-bench-benchmarking-multimodal-large","title":"SEED-Bench: Benchmarking Multimodal Large Language Models","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/advancing-ttp-analysis-harnessing-the-power","slug":"advancing-ttp-analysis-harnessing-the-power","title":"Advancing TTP Analysis: Harnessing the Power of Large Language Models with Retrieval Augmented Generation","date":"2023-12-30","arxiv_id":"2401.00280","n_code_links":1,"syntology":null},{"paper":null,"slug":"trace-and-edit-relation-associations-in-gpt","title":"Trace and Edit Relation Associations in GPT","date":"2023-12-30","arxiv_id":"2401.02976","n_code_links":0,"syntology":null},{"paper":"/paper/gemini-in-reasoning-unveiling-commonsense-in","slug":"gemini-in-reasoning-unveiling-commonsense-in","title":"Gemini in Reasoning: Unveiling Commonsense in Multimodal Large Language Models","date":"2023-12-29","arxiv_id":"2312.17661","n_code_links":1,"syntology":null},{"paper":"/paper/jatmo-prompt-injection-defense-by-task","slug":"jatmo-prompt-injection-defense-by-task","title":"Jatmo: Prompt Injection Defense by Task-Specific Finetuning","date":"2023-12-29","arxiv_id":"2312.17673","n_code_links":1,"syntology":{"ran":10,"of":18,"n_ran_checked":10,"n_instrument":0,"unverified":8,"pointer_only":18,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","official":{"repos":["wagner-group/prompt-injection-defense"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-the-performance-of-large-language-1","title":"Evaluating the Performance of Large Language Models for Spanish Language in Undergraduate Admissions Exams","date":"2023-12-28","arxiv_id":"2312.16845","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-model-for-causal-decision","title":"LLM4Causal: Democratized Causal Tools for Everyone via Large Language Model","date":"2023-12-28","arxiv_id":"2312.17122","n_code_links":0,"syntology":null},{"paper":null,"slug":"pangu-p-enhancing-language-model","title":"PanGu-$π$: Enhancing Language Model Architectures via Nonlinearity Compensation","date":"2023-12-27","arxiv_id":"2312.17276","n_code_links":0,"syntology":null},{"paper":null,"slug":"chartbench-a-benchmark-for-complex-visual","title":"ChartBench: A Benchmark for Complex Visual Reasoning in Charts","date":"2023-12-26","arxiv_id":"2312.15915","n_code_links":0,"syntology":null},{"paper":"/paper/principled-instructions-are-all-you-need-for","slug":"principled-instructions-are-all-you-need-for","title":"Principled Instructions Are All You Need for Questioning LLaMA-1/2, GPT-3.5/4","date":"2023-12-26","arxiv_id":"2312.16171","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vila-lab/atlas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/secqa-a-concise-question-answering-dataset","slug":"secqa-a-concise-question-answering-dataset","title":"SecQA: A Concise Question-Answering Dataset for Evaluating Large Language Models in Computer Security","date":"2023-12-26","arxiv_id":"2312.15838","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zefang-liu/lm-evaluation-harness"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"task-contamination-language-models-may-not-be","title":"Task Contamination: Language Models May Not Be Few-Shot Anymore","date":"2023-12-26","arxiv_id":"2312.16337","n_code_links":0,"syntology":null},{"paper":"/paper/fairness-aware-structured-pruning-in","slug":"fairness-aware-structured-pruning-in","title":"Fairness-Aware Structured Pruning in Transformers","date":"2023-12-24","arxiv_id":"2312.15398","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["chandar-lab/fasp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"do-llm-agents-exhibit-social-behavior","title":"Do LLM Agents Exhibit Social Behavior?","date":"2023-12-23","arxiv_id":"2312.15198","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-the-potential-of-fpga-based","slug":"understanding-the-potential-of-fpga-based","title":"Understanding the Potential of FPGA-Based Spatial Acceleration for Large Language Model Inference","date":"2023-12-23","arxiv_id":"2312.15159","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficacy-of-machine-generated-instructions","title":"Efficacy of Machine-Generated Instructions","date":"2023-12-22","arxiv_id":"2312.14423","n_code_links":0,"syntology":null},{"paper":null,"slug":"fm-ov3d-foundation-model-based-cross-modal","title":"FM-OV3D: Foundation Model-based Cross-modal Knowledge Blending for Open-Vocabulary 3D Detection","date":"2023-12-22","arxiv_id":"2312.14465","n_code_links":0,"syntology":null},{"paper":"/paper/refining-gpt-3-embeddings-with-a-siamese","slug":"refining-gpt-3-embeddings-with-a-siamese","title":"Refining GPT-3 Embeddings with a Siamese Structure for Technical Post Duplicate Detection","date":"2023-12-22","arxiv_id":"2312.15068","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-the-regularity-of-self","title":"How Smooth Is Attention?","date":"2023-12-22","arxiv_id":"2312.14820","n_code_links":0,"syntology":null},{"paper":"/paper/argue-with-me-tersely-towards-sentence-level","slug":"argue-with-me-tersely-towards-sentence-level","title":"Argue with Me Tersely: Towards Sentence-Level Counter-Argument Generation","date":"2023-12-21","arxiv_id":"2312.13608","n_code_links":1,"syntology":null},{"paper":"/paper/chatgpt-as-a-commenter-to-the-news-can-llms","slug":"chatgpt-as-a-commenter-to-the-news-can-llms","title":"ChatGPT as a commenter to the news: can LLMs generate human-like opinions?","date":"2023-12-21","arxiv_id":"2312.13961","n_code_links":1,"syntology":null},{"paper":"/paper/de-novo-drug-design-using-reinforcement-1","slug":"de-novo-drug-design-using-reinforcement-1","title":"De novo Drug Design using Reinforcement Learning with Multiple GPT Agents","date":"2023-12-21","arxiv_id":"2401.06155","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hxyfighter/molrl-mgpt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"infovisdial-an-informative-visual-dialogue","title":"InfoVisDial: An Informative Visual Dialogue Dataset by Bridging Large Multimodal and Language Models","date":"2023-12-21","arxiv_id":"2312.13503","n_code_links":0,"syntology":null},{"paper":"/paper/provfl-client-driven-interpretability-of","slug":"provfl-client-driven-interpretability-of","title":"TraceFL: Interpretability-Driven Debugging in Federated Learning via Neuron Provenance","date":"2023-12-21","arxiv_id":"2312.13632","n_code_links":2,"syntology":null},{"paper":null,"slug":"team-irisapu-project-description-for-drc2023","title":"Team Irisapu Project Description for DRC2023","date":"2023-12-21","arxiv_id":"2312.13765","n_code_links":0,"syntology":null},{"paper":null,"slug":"typhoon-thai-large-language-models","title":"Typhoon: Thai Large Language Models","date":"2023-12-21","arxiv_id":"2312.13951","n_code_links":0,"syntology":null},{"paper":"/paper/agentcoder-multi-agent-based-code-generation","slug":"agentcoder-multi-agent-based-code-generation","title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","date":"2023-12-20","arxiv_id":"2312.13010","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["huangd1999/AgentCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"benchmarking-and-analyzing-in-context","title":"Benchmarking and Analyzing In-context Learning, Fine-tuning and Supervised Learning for Biomedical Knowledge Curation: a focused study on chemical entities of biological interest","date":"2023-12-20","arxiv_id":"2312.12989","n_code_links":0,"syntology":null},{"paper":"/paper/domain-specific-code-language-models","slug":"domain-specific-code-language-models","title":"MonoCoder: Domain-Specific Code Language Model for HPC Codes and Tasks","date":"2023-12-20","arxiv_id":"2312.13322","n_code_links":3,"syntology":null},{"paper":null,"slug":"can-chatgpt-be-your-personal-medical","title":"Can ChatGPT be Your Personal Medical Assistant?","date":"2023-12-19","arxiv_id":"2312.12006","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-transformers-learn-sequential-function","title":"Can Transformers Learn Sequential Function Classes In Context?","date":"2023-12-19","arxiv_id":"2312.12655","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-in-medical-term","title":"Large Language Models in Medical Term Classification and Unexpected Misalignment Between Response and Reasoning","date":"2023-12-19","arxiv_id":"2312.14184","n_code_links":0,"syntology":null},{"paper":"/paper/an-in-depth-look-at-gemini-s-language","slug":"an-in-depth-look-at-gemini-s-language","title":"An In-depth Look at Gemini's Language Abilities","date":"2023-12-18","arxiv_id":"2312.11444","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":11,"n_instrument":0,"unverified":1,"pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["neulab/gemini-benchmark"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generative-linguistic-representation-for","title":"Generative linguistic representation for spoken language identification","date":"2023-12-18","arxiv_id":"2312.10964","n_code_links":0,"syntology":null},{"paper":null,"slug":"artificial-intelligence-optical-hardware","title":"Artificial intelligence optical hardware empowers high-resolution hyperspectral video understanding at 1.2 Tb/s","date":"2023-12-17","arxiv_id":"2312.10639","n_code_links":0,"syntology":null},{"paper":"/paper/decoding-concerns-multi-label-classification","slug":"decoding-concerns-multi-label-classification","title":"Decoding Concerns: Multi-label Classification of Vaccine Sentiments in Social Media","date":"2023-12-17","arxiv_id":"2312.10626","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-ai-vocational-skills-through","title":"Evaluating AI Vocational Skills Through Professional Testing","date":"2023-12-17","arxiv_id":"2312.10603","n_code_links":0,"syntology":null},{"paper":"/paper/hyperpie-hyperparameter-information","slug":"hyperpie-hyperparameter-information","title":"HyperPIE: Hyperparameter Information Extraction from Scientific Publications","date":"2023-12-17","arxiv_id":"2312.10638","n_code_links":1,"syntology":null},{"paper":null,"slug":"mixed-distillation-helps-smaller-language","title":"Mixed Distillation Helps Smaller Language Model Better Reasoning","date":"2023-12-17","arxiv_id":"2312.10730","n_code_links":0,"syntology":null},{"paper":"/paper/multi-label-classification-of-covid-tweets","slug":"multi-label-classification-of-covid-tweets","title":"Multi-Label Classification of COVID-Tweets Using Large Language Models","date":"2023-12-17","arxiv_id":"2312.10748","n_code_links":1,"syntology":null},{"paper":null,"slug":"t2m-hifigpt-generating-high-quality-human","title":"T2M-HiFiGPT: Generating High Quality Human Motion from Textual Descriptions with Residual Discrete Representations","date":"2023-12-17","arxiv_id":"2312.10628","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-analysis-of-large-language","title":"A Comparative Analysis of Large Language Models for Code Documentation Generation","date":"2023-12-16","arxiv_id":"2312.10349","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-novel-dataset-for-financial-education-text","title":"A Novel Dataset for Financial Education Text Simplification in Spanish","date":"2023-12-15","arxiv_id":"2312.09897","n_code_links":0,"syntology":null},{"paper":null,"slug":"distilling-large-language-models-for-matching","title":"Distilling Large Language Models for Matching Patients to Clinical Trials","date":"2023-12-15","arxiv_id":"2312.09958","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-multi-level-threats-in-telegram","slug":"exploring-multi-level-threats-in-telegram","title":"Exploring Multi-Level Threats in Telegram Data with AI-Human Annotation: A Preliminary Study","date":"2023-12-15","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"red-ai-inconsistent-responses-from-gpt3-5","title":"Red AI? Inconsistent Responses from GPT3.5 Models on Political Issues in the US and China","date":"2023-12-15","arxiv_id":"2312.09917","n_code_links":0,"syntology":null},{"paper":"/paper/boosting-llm-reasoning-push-the-limits-of-few","slug":"boosting-llm-reasoning-push-the-limits-of-few","title":"Fewer is More: Boosting LLM Reasoning with Reinforced Context Pruning","date":"2023-12-14","arxiv_id":"2312.08901","n_code_links":0,"syntology":null},{"paper":null,"slug":"entity-augmented-code-generation","title":"Dynamic Retrieval-Augmented Generation","date":"2023-12-14","arxiv_id":"2312.08976","n_code_links":0,"syntology":null},{"paper":null,"slug":"inter-layer-scheduling-space-exploration-for","title":"Inter-Layer Scheduling Space Exploration for Multi-model Inference on Heterogeneous Chiplets","date":"2023-12-14","arxiv_id":"2312.09401","n_code_links":0,"syntology":null},{"paper":null,"slug":"motion-flow-matching-for-human-motion","title":"Motion Flow Matching for Human Motion Synthesis and Editing","date":"2023-12-14","arxiv_id":"2312.08895","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-evaluation-improves-selective-generation","title":"Self-Evaluation Improves Selective Generation in Large Language Models","date":"2023-12-14","arxiv_id":"2312.09300","n_code_links":0,"syntology":null},{"paper":null,"slug":"successor-heads-recurring-interpretable","title":"Successor Heads: Recurring, Interpretable Attention Heads In The Wild","date":"2023-12-14","arxiv_id":"2312.09230","n_code_links":0,"syntology":null},{"paper":"/paper/tinygsm-achieving-80-on-gsm8k-with-small","slug":"tinygsm-achieving-80-on-gsm8k-with-small","title":"TinyGSM: achieving >80% on GSM8k with small language models","date":"2023-12-14","arxiv_id":"2312.09241","n_code_links":0,"syntology":null},{"paper":null,"slug":"weak-to-strong-generalization-eliciting","title":"Weak-to-Strong Generalization: Eliciting Strong Capabilities With Weak Supervision","date":"2023-12-14","arxiv_id":"2312.09390","n_code_links":0,"syntology":null},{"paper":"/paper/weaving-pathways-for-justice-with-gpt-llm","slug":"weaving-pathways-for-justice-with-gpt-llm","title":"Weaving Pathways for Justice with GPT: LLM-driven automated drafting of interactive legal applications","date":"2023-12-14","arxiv_id":"2312.09198","n_code_links":1,"syntology":null},{"paper":"/paper/causality-analysis-for-evaluating-the","slug":"causality-analysis-for-evaluating-the","title":"Causality Analysis for Evaluating the Security of Large Language Models","date":"2023-12-13","arxiv_id":"2312.07876","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":6,"n_instrument":1,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["casperllm/casper"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-language-models-are-complex-table","title":"Large Language Models are Complex Table Parsers","date":"2023-12-13","arxiv_id":"2312.11521","n_code_links":0,"syntology":null},{"paper":null,"slug":"native-language-identification-with-large","title":"Native Language Identification with Large Language Models","date":"2023-12-13","arxiv_id":"2312.07819","n_code_links":0,"syntology":null},{"paper":"/paper/ai-control-improving-safety-despite","slug":"ai-control-improving-safety-despite","title":"AI Control: Improving Safety Despite Intentional Subversion","date":"2023-12-12","arxiv_id":"2312.06942","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rgreenblatt/control-evaluations"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-large-language-models-to-facilitate","title":"Exploring Large Language Models to Facilitate Variable Autonomy for Human-Robot Teaming","date":"2023-12-12","arxiv_id":"2312.07214","n_code_links":0,"syntology":null},{"paper":"/paper/image-content-generation-with-causal","slug":"image-content-generation-with-causal","title":"Image Content Generation with Causal Reasoning","date":"2023-12-12","arxiv_id":"2312.07132","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":2,"n_instrument":2,"unverified":4,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ieit-agi/mix-shannon"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/multilingual-large-language-models-leak-human","slug":"multilingual-large-language-models-leak-human","title":"Multilingual large language models leak human stereotypes across language boundaries","date":"2023-12-12","arxiv_id":"2312.07141","n_code_links":1,"syntology":null},{"paper":"/paper/perseus-removing-energy-bloat-from-large","slug":"perseus-removing-energy-bloat-from-large","title":"Reducing Energy Bloat in Large Model Training","date":"2023-12-12","arxiv_id":"2312.06902","n_code_links":2,"syntology":null},{"paper":null,"slug":"sm70-a-large-language-model-for-medical","title":"SM70: A Large Language Model for Medical Devices","date":"2023-12-12","arxiv_id":"2312.06974","n_code_links":0,"syntology":null},{"paper":"/paper/can-it-edit-evaluating-the-ability-of-large","slug":"can-it-edit-evaluating-the-ability-of-large","title":"Can It Edit? Evaluating the Ability of Large Language Models to Follow Code Editing Instructions","date":"2023-12-11","arxiv_id":"2312.12450","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["nuprl/canitedit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-chatgpt-as-a-question-answering","title":"Evaluating ChatGPT as a Question Answering System: A Comprehensive Analysis and Comparison with Existing Models","date":"2023-12-11","arxiv_id":"2312.07592","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-large-language-models-are-all","title":"Generative Large Language Models Are All-purpose Text Analytics Engines: Text-to-text Learning Is All Your Need","date":"2023-12-11","arxiv_id":"2312.06099","n_code_links":0,"syntology":null},{"paper":null,"slug":"survey-on-foundation-models-for-prognostics","title":"Survey on Foundation Models for Prognostics and Health Management in Industrial Cyber-Physical Systems","date":"2023-12-11","arxiv_id":"2312.06261","n_code_links":0,"syntology":null},{"paper":null,"slug":"early-chatgpt-user-portrait-through-the-lens","title":"Early ChatGPT User Portrait through the Lens of Data","date":"2023-12-10","arxiv_id":"2312.10078","n_code_links":0,"syntology":null},{"paper":"/paper/sim-gpt-text-similarity-via-gpt-annotated","slug":"sim-gpt-text-similarity-via-gpt-annotated","title":"Sim-GPT: Text Similarity via GPT Annotated Data","date":"2023-12-09","arxiv_id":"2312.05603","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-the-limits-of-chatgpt-in-software","title":"Exploring the Limits of ChatGPT in Software Security Applications","date":"2023-12-08","arxiv_id":"2312.05275","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-interactive-optimization-of-open-source","title":"LLM Interactive Optimization of Open Source Python Libraries -- Case Studies and Generalization","date":"2023-12-08","arxiv_id":"2312.14949","n_code_links":0,"syntology":null},{"paper":null,"slug":"make-them-spill-the-beans-coercive-knowledge","title":"Make Them Spill the Beans! Coercive Knowledge Extraction from (Production) LLMs","date":"2023-12-08","arxiv_id":"2312.04782","n_code_links":0,"syntology":null}],"record_sha256":"428769dec9c3155835068b39b60a63f9713c01a2a2899f6e9f4169abc759a8ea","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}