{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-cosine-annealing/papers/12","list_of":"/method/linear-warmup-with-cosine-annealing","method":"Linear Warmup With Cosine Annealing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":12,"pages_in_order":38,"rows_per_page":100,"rows":[1101,1200],"of":3797,"counts":{"archive_papers_tagged":3797,"with_a_code_link":1655,"where_syntology_ran_a_sample":602,"not_listed_spam_title":0,"listed":3797,"listed_where_code_ran":602,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":490,"every_run_a_failure_of_syntologys_instrument":112,"listed_with_a_run_with_no_instrument_failure":490,"listed_every_run_a_failure_of_syntologys_instrument":112,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-cosine-annealing","prev":"/method/linear-warmup-with-cosine-annealing/papers/11","next":"/method/linear-warmup-with-cosine-annealing/papers/13","papers":[{"paper":"/paper/evaluating-mathematical-reasoning-of-large","slug":"evaluating-mathematical-reasoning-of-large","title":"Evaluating Mathematical Reasoning of Large Language Models: A Focus on Error Identification and Correction","date":"2024-06-02","arxiv_id":"2406.00755","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["littlecirc1e/eic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"focus-forging-originality-through-contrastive","title":"FOCUS: Forging Originality through Contrastive Use in Self-Plagiarism for Language Models","date":"2024-06-02","arxiv_id":"2406.00839","n_code_links":0,"syntology":null},{"paper":null,"slug":"pretrained-hybrids-with-mad-skills","title":"Pretrained Hybrids with MAD Skills","date":"2024-06-02","arxiv_id":"2406.00894","n_code_links":0,"syntology":null},{"paper":null,"slug":"2406-07572","title":"Domain-specific ReAct for physics-integrated iterative modeling: A case study of LLM agents for gas path analysis of gas turbines","date":"2024-06-01","arxiv_id":"2406.07572","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-evaluation-benchmark-for-autoformalization","title":"An Evaluation Benchmark for Autoformalization in Lean4","date":"2024-06-01","arxiv_id":"2406.06555","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-metrics-evaluating-llms-effectiveness","title":"Beyond Metrics: Evaluating LLMs' Effectiveness in Culturally Nuanced, Low-Resource Real-World Scenarios","date":"2024-06-01","arxiv_id":"2406.00343","n_code_links":0,"syntology":null},{"paper":"/paper/generative-ai-voting-fair-collective-choice","slug":"generative-ai-voting-fair-collective-choice","title":"Generative AI Voting: Fair Collective Choice is Resilient to LLM Biases and Inconsistencies","date":"2024-05-31","arxiv_id":"2406.11871","n_code_links":1,"syntology":null},{"paper":"/paper/hard-cases-detection-in-motion-prediction-by","slug":"hard-cases-detection-in-motion-prediction-by","title":"Hard Cases Detection in Motion Prediction by Vision-Language Foundation Models","date":"2024-05-31","arxiv_id":"2405.20991","n_code_links":1,"syntology":null},{"paper":"/paper/large-language-models-are-zero-shot-next","slug":"large-language-models-are-zero-shot-next","title":"Large Language Models are Zero-Shot Next Location Predictors","date":"2024-05-31","arxiv_id":"2405.20962","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ssai-trento/llm-zero-shot-nl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lolameme-logic-language-memory-mechanistic","title":"LOLAMEME: Logic, Language, Memory, Mechanistic Framework","date":"2024-05-31","arxiv_id":"2406.02592","n_code_links":0,"syntology":null},{"paper":"/paper/multilingual-text-style-transfer-datasets","slug":"multilingual-text-style-transfer-datasets","title":"Multilingual Text Style Transfer: Datasets & Models for Indian Languages","date":"2024-05-31","arxiv_id":"2405.20805","n_code_links":2,"syntology":null},{"paper":null,"slug":"the-point-of-view-of-a-sentiment-towards","title":"The Point of View of a Sentiment: Towards Clinician Bias Detection in Psychiatric Notes","date":"2024-05-31","arxiv_id":"2405.20582","n_code_links":0,"syntology":null},{"paper":"/paper/anah-analytical-annotation-of-hallucinations","slug":"anah-analytical-annotation-of-hallucinations","title":"ANAH: Analytical Annotation of Hallucinations in Large Language Models","date":"2024-05-30","arxiv_id":"2405.20315","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["open-compass/anah"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":null,"slug":"autobreach-universal-and-adaptive","title":"AutoBreach: Universal and Adaptive Jailbreaking with Efficient Wordplay-Guided Optimization","date":"2024-05-30","arxiv_id":"2405.19668","n_code_links":0,"syntology":null},{"paper":null,"slug":"divide-and-conquer-meets-consensus-unleashing","title":"Divide-and-Conquer Meets Consensus: Unleashing the Power of Functions in Code Generation","date":"2024-05-30","arxiv_id":"2405.20092","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-graph-tuning-real-time-large","title":"Knowledge Graph Tuning: Real-time Large Language Model Personalization based on Human Feedback","date":"2024-05-30","arxiv_id":"2405.19686","n_code_links":0,"syntology":null},{"paper":"/paper/llamea-a-large-language-model-evolutionary","slug":"llamea-a-large-language-model-evolutionary","title":"LLaMEA: A Large Language Model Evolutionary Algorithm for Automatically Generating Metaheuristics","date":"2024-05-30","arxiv_id":"2405.20132","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nikivanstein/LLaMEA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"phantom-general-trigger-attacks-on-retrieval","title":"Phantom: General Trigger Attacks on Retrieval Augmented Language Generation","date":"2024-05-30","arxiv_id":"2405.20485","n_code_links":0,"syntology":null},{"paper":null,"slug":"robo-instruct-simulator-augmented-instruction","title":"Robo-Instruct: Simulator-Augmented Instruction Alignment For Finetuning Code LLMs","date":"2024-05-30","arxiv_id":"2405.20179","n_code_links":0,"syntology":null},{"paper":null,"slug":"significance-of-chain-of-thought-in-gender","title":"Significance of Chain of Thought in Gender Bias Mitigation for English-Dravidian Machine Translation","date":"2024-05-30","arxiv_id":"2405.19701","n_code_links":0,"syntology":null},{"paper":"/paper/towards-ontology-enhanced-representation","slug":"towards-ontology-enhanced-representation","title":"Towards Ontology-Enhanced Representation Learning for Large Language Models","date":"2024-05-30","arxiv_id":"2405.20527","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-multi-source-retrieval-question-answering","title":"A Multi-Source Retrieval Question Answering Framework Based on RAG","date":"2024-05-29","arxiv_id":"2405.19207","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-agreement-diagnosing-the-rationale","slug":"beyond-agreement-diagnosing-the-rationale","title":"Beyond Agreement: Diagnosing the Rationale Alignment of Automated Essay Scoring Methods based on Linguistically-informed Counterfactuals","date":"2024-05-29","arxiv_id":"2405.19433","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-gpt-redefine-medical-understanding","title":"Can GPT Redefine Medical Understanding? Evaluating GPT on Biomedical Machine Reading Comprehension","date":"2024-05-29","arxiv_id":"2405.18682","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-model-agnostic-alignment-via","title":"Efficient Model-agnostic Alignment via Bayesian Persuasion","date":"2024-05-29","arxiv_id":"2405.18718","n_code_links":0,"syntology":null},{"paper":null,"slug":"lmo-dp-optimizing-the-randomization-mechanism","title":"LMO-DP: Optimizing the Randomization Mechanism for Differentially Private Fine-Tuning (Large) Language Models","date":"2024-05-29","arxiv_id":"2405.18776","n_code_links":0,"syntology":null},{"paper":"/paper/map-neo-highly-capable-and-transparent","slug":"map-neo-highly-capable-and-transparent","title":"MAP-Neo: Highly Capable and Transparent Bilingual Large Language Model Series","date":"2024-05-29","arxiv_id":"2405.19327","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":12,"n_instrument":0,"unverified":2,"pointer_only":14,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["multimodal-art-projection/map-neo"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/aligning-to-thousands-of-preferences-via","slug":"aligning-to-thousands-of-preferences-via","title":"Aligning to Thousands of Preferences via System Message Generalization","date":"2024-05-28","arxiv_id":"2405.17977","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kaistAI/Janus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"paper":"/paper/an-empirical-analysis-on-large-language","slug":"an-empirical-analysis-on-large-language","title":"An Empirical Analysis on Large Language Models in Debate Evaluation","date":"2024-05-28","arxiv_id":"2406.00050","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["xinyiliu0227/llm_debate_bias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"are-ppo-ed-language-models-hackable","title":"Are PPO-ed Language Models Hackable?","date":"2024-05-28","arxiv_id":"2406.02577","n_code_links":0,"syntology":null},{"paper":null,"slug":"edinburgh-clinical-nlp-at-mediqa-corr-2024","title":"Edinburgh Clinical NLP at MEDIQA-CORR 2024: Guiding Large Language Models with Hints","date":"2024-05-28","arxiv_id":"2405.18028","n_code_links":0,"syntology":null},{"paper":"/paper/llms-and-memorization-on-quality-and","slug":"llms-and-memorization-on-quality-and","title":"LLMs and Memorization: On Quality and Specificity of Copyright Compliance","date":"2024-05-28","arxiv_id":"2405.18492","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-intrinsic-socioeconomic-biases","title":"Understanding Intrinsic Socioeconomic Biases in Large Language Models","date":"2024-05-28","arxiv_id":"2405.18662","n_code_links":0,"syntology":null},{"paper":"/paper/assessing-llms-suitability-for-knowledge","slug":"assessing-llms-suitability-for-knowledge","title":"Assessing LLMs Suitability for Knowledge Graph Completion","date":"2024-05-27","arxiv_id":"2405.17249","n_code_links":1,"syntology":null},{"paper":"/paper/inversionview-a-general-purpose-method-for","slug":"inversionview-a-general-purpose-method-for","title":"InversionView: A General-Purpose Method for Reading Information from Neural Activations","date":"2024-05-27","arxiv_id":"2405.17653","n_code_links":1,"syntology":{"ran":1,"of":6,"n_ran_checked":1,"n_instrument":0,"unverified":5,"pointer_only":6,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["huangxt39/inversionview"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"llm-based-cooperative-agents-using","title":"REVECA: Adaptive Planning and Trajectory-based Validation in Cooperative Language Agents using Information Relevance and Relative Proximity","date":"2024-05-27","arxiv_id":"2405.16751","n_code_links":0,"syntology":null},{"paper":null,"slug":"performance-evaluation-of-reddit-comments","title":"Performance evaluation of Reddit Comments using Machine Learning and Natural Language Processing methods in Sentiment Analysis","date":"2024-05-27","arxiv_id":"2405.16810","n_code_links":0,"syntology":null},{"paper":"/paper/reflectioncoder-learning-from-reflection","slug":"reflectioncoder-learning-from-reflection","title":"ReflectionCoder: Learning from Reflection Sequence for Enhanced One-off Code Generation","date":"2024-05-27","arxiv_id":"2405.17057","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["sensellm/reflectioncoder"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/rtl-repo-a-benchmark-for-evaluating-llms-on","slug":"rtl-repo-a-benchmark-for-evaluating-llms-on","title":"RTL-Repo: A Benchmark for Evaluating LLMs on Large-Scale RTL Design Projects","date":"2024-05-27","arxiv_id":"2405.17378","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-scaling-law-in-stellar-light-curves","title":"The Scaling Law in Stellar Light Curves","date":"2024-05-27","arxiv_id":"2405.17156","n_code_links":0,"syntology":null},{"paper":"/paper/thread-thinking-deeper-with-recursive","slug":"thread-thinking-deeper-with-recursive","title":"THREAD: Thinking Deeper with Recursive Spawning","date":"2024-05-27","arxiv_id":"2405.17402","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["philipmit/thread"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/accelerating-transformers-with-spectrum-1","slug":"accelerating-transformers-with-spectrum-1","title":"Accelerating Transformers with Spectrum-Preserving Token Merging","date":"2024-05-25","arxiv_id":"2405.16148","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":2,"n_instrument":4,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hchautran/PiToMe"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/automanual-generating-instruction-manuals-by","slug":"automanual-generating-instruction-manuals-by","title":"AutoManual: Constructing Instruction Manuals by LLM Agents via Interactive Environmental Learning","date":"2024-05-25","arxiv_id":"2405.16247","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["minghchen/automanual"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"incremental-comprehension-of-garden-path","title":"Incremental Comprehension of Garden-Path Sentences by Large Language Models: Semantic Interpretation, Syntactic Re-Analysis, and Attention","date":"2024-05-25","arxiv_id":"2405.16042","n_code_links":0,"syntology":null},{"paper":null,"slug":"mindstar-enhancing-math-reasoning-in-pre","title":"MindStar: Enhancing Math Reasoning in Pre-trained LLMs at Inference Time","date":"2024-05-25","arxiv_id":"2405.16265","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-evaluation-of-estimative-uncertainty-in","title":"An Evaluation of Estimative Uncertainty in Large Language Models","date":"2024-05-24","arxiv_id":"2405.15185","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-pre-trained-large-language","title":"Benchmarking the Performance of Pre-trained LLMs across Urdu NLP Tasks","date":"2024-05-24","arxiv_id":"2405.15453","n_code_links":0,"syntology":null},{"paper":"/paper/culturepark-boosting-cross-cultural","slug":"culturepark-boosting-cross-cultural","title":"CulturePark: Boosting Cross-cultural Understanding in Large Language Models","date":"2024-05-24","arxiv_id":"2405.15145","n_code_links":1,"syntology":{"ran":0,"of":7,"n_ran_checked":0,"n_instrument":0,"unverified":7,"pointer_only":7,"phrase":"0 ran · 7 unverified","official":{"repos":["scarelette/culturepark"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":[]}}},{"paper":"/paper/evaluating-the-adversarial-robustness-of-1","slug":"evaluating-the-adversarial-robustness-of-1","title":"Evaluating and Safeguarding the Adversarial Robustness of Retrieval-Based In-Context Learning","date":"2024-05-24","arxiv_id":"2405.15984","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["simonucl/adv-retreival-icl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/generalizable-and-scalable-multistage","slug":"generalizable-and-scalable-multistage","title":"Generalizable and Scalable Multistage Biomedical Concept Normalization Leveraging Large Language Models","date":"2024-05-24","arxiv_id":"2405.15122","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-is-not-an-annotator-the-necessity-of","title":"GPT is Not an Annotator: The Necessity of Human Annotation in Fairness Benchmark Construction","date":"2024-05-24","arxiv_id":"2405.15760","n_code_links":0,"syntology":null},{"paper":"/paper/learning-the-language-of-protein-structure","slug":"learning-the-language-of-protein-structure","title":"Learning the Language of Protein Structure","date":"2024-05-24","arxiv_id":"2405.15840","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["instadeepai/protein-structure-tokenizer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-impact-and-opportunities-of-generative-ai","title":"The Impact and Opportunities of Generative AI in Fact-Checking","date":"2024-05-24","arxiv_id":"2405.15985","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-understanding-how-transformer-perform","title":"The Buffer Mechanism for Multi-Step Information Reasoning in Language Models","date":"2024-05-24","arxiv_id":"2405.15302","n_code_links":0,"syntology":null},{"paper":"/paper/editworld-simulating-world-dynamics-for","slug":"editworld-simulating-world-dynamics-for","title":"EditWorld: Simulating World Dynamics for Instruction-Following Image Editing","date":"2024-05-23","arxiv_id":"2405.14785","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yangling0818/editworld"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/eliciting-informative-text-evaluations-with","slug":"eliciting-informative-text-evaluations-with","title":"Eliciting Informative Text Evaluations with Large Language Models","date":"2024-05-23","arxiv_id":"2405.15077","n_code_links":1,"syntology":{"ran":10,"of":14,"n_ran_checked":8,"n_instrument":2,"unverified":4,"pointer_only":14,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yx-lu/eliciting-informative-text-evaluations-with-large-language-models"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/from-explicit-cot-to-implicit-cot-learning-to","slug":"from-explicit-cot-to-implicit-cot-learning-to","title":"From Explicit CoT to Implicit CoT: Learning to Internalize CoT Step by Step","date":"2024-05-23","arxiv_id":"2405.14838","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["da03/internalize_cot_step_by_step"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-language-models-can-self-correct-with","title":"Large Language Models Can Self-Correct with Key Condition Verification","date":"2024-05-23","arxiv_id":"2405.14092","n_code_links":0,"syntology":null},{"paper":"/paper/not-all-language-model-features-are-linear","slug":"not-all-language-model-features-are-linear","title":"Not All Language Model Features Are Linear","date":"2024-05-23","arxiv_id":"2405.14860","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["joshengels/multidimensionalfeatures"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/wise-rethinking-the-knowledge-memory-for","slug":"wise-rethinking-the-knowledge-memory-for","title":"WISE: Rethinking the Knowledge Memory for Lifelong Model Editing of Large Language Models","date":"2024-05-23","arxiv_id":"2405.14768","n_code_links":1,"syntology":{"ran":13,"of":16,"n_ran_checked":4,"n_instrument":9,"unverified":3,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 9 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zjunlp/easyedit"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"automatically-identifying-local-and-global","title":"Automatically Identifying Local and Global Circuits with Linear Computation Graphs","date":"2024-05-22","arxiv_id":"2405.13868","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-large-language-models-with-human","slug":"evaluating-large-language-models-with-human","title":"Evaluating Large Language Models with Human Feedback: Establishing a Swedish Benchmark","date":"2024-05-22","arxiv_id":"2405.14006","n_code_links":1,"syntology":null},{"paper":null,"slug":"ku-dmis-at-ehrsql-2024-generating-sql-query","title":"KU-DMIS at EHRSQL 2024:Generating SQL query via question templatization in EHR","date":"2024-05-22","arxiv_id":"2406.00014","n_code_links":0,"syntology":null},{"paper":"/paper/topa-extend-large-language-models-for-video","slug":"topa-extend-large-language-models-for-video","title":"TOPA: Extending Large Language Models for Video Understanding via Text-Only Pre-Alignment","date":"2024-05-22","arxiv_id":"2405.13911","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":4,"n_instrument":2,"unverified":3,"pointer_only":3,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["dhg-wei/topa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"generative-ai-and-large-language-models-for","title":"Generative AI in Cybersecurity: A Comprehensive Review of LLM Applications and Vulnerabilities","date":"2024-05-21","arxiv_id":"2405.12750","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-reliable-ai-chatbots-are-for-disease","title":"How Reliable AI Chatbots are for Disease Prediction from Patient Complaints?","date":"2024-05-21","arxiv_id":"2405.13219","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-persuasion-techniques-in-arabic","title":"Investigating Persuasion Techniques in Arabic: An Empirical Study Leveraging Large Language Models","date":"2024-05-21","arxiv_id":"2405.12884","n_code_links":0,"syntology":null},{"paper":"/paper/quantifying-emergence-in-large-language","slug":"quantifying-emergence-in-large-language","title":"Quantifying Semantic Emergence in Language Models","date":"2024-05-21","arxiv_id":"2405.12617","n_code_links":1,"syntology":{"ran":1,"of":6,"n_ran_checked":1,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["zodiark-ch/emergence-of-llms"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-and-modeling-social-intelligence-a","slug":"evaluating-and-modeling-social-intelligence-a","title":"Evaluating and Modeling Social Intelligence: A Comparative Study of Human and AI Capabilities","date":"2024-05-20","arxiv_id":"2405.11841","n_code_links":1,"syntology":null},{"paper":null,"slug":"davinci-at-semeval-2024-task-9-few-shot","title":"DaVinci at SemEval-2024 Task 9: Few-shot prompting GPT-3.5 for Unconventional Reasoning","date":"2024-05-19","arxiv_id":"2405.11559","n_code_links":0,"syntology":null},{"paper":"/paper/human-centered-llm-agent-user-interface-a","slug":"human-centered-llm-agent-user-interface-a","title":"Human-Centered LLM-Agent User Interface: A Position Paper","date":"2024-05-19","arxiv_id":"2405.13050","n_code_links":1,"syntology":null},{"paper":"/paper/your-transformer-is-secretly-linear","slug":"your-transformer-is-secretly-linear","title":"Your Transformer is Secretly Linear","date":"2024-05-19","arxiv_id":"2405.12250","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["AIRI-Institute/LLM-Microscope"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/zero-shot-stance-detection-using-contextual","slug":"zero-shot-stance-detection-using-contextual","title":"Zero-Shot Stance Detection using Contextual Data Generation with LLMs","date":"2024-05-19","arxiv_id":"2405.11637","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Babakbehkamkia/GPT-Stance-Detection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cross-language-assessment-of-mathematical","title":"Cross-Language Assessment of Mathematical Capability of ChatGPT","date":"2024-05-18","arxiv_id":"2405.11264","n_code_links":0,"syntology":null},{"paper":"/paper/evaluation-of-large-language-model","slug":"evaluation-of-large-language-model","title":"Evaluation of large language model performance on the Biomedical Language Understanding and Reasoning Benchmark","date":"2024-05-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/language-models-can-exploit-cross-task-in","slug":"language-models-can-exploit-cross-task-in","title":"Language Models can Exploit Cross-Task In-context Learning for Data-Scarce Novel Tasks","date":"2024-05-17","arxiv_id":"2405.10548","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["c-anwoy/cross-task-icl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fintextqa-a-dataset-for-long-form-financial","title":"FinTextQA: A Dataset for Long-form Financial Question Answering","date":"2024-05-16","arxiv_id":"2405.09980","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-store-mining-and-analysis","title":"GPT Store Mining and Analysis","date":"2024-05-16","arxiv_id":"2405.10210","n_code_links":0,"syntology":null},{"paper":"/paper/hw-gpt-bench-hardware-aware-architecture","slug":"hw-gpt-bench-hardware-aware-architecture","title":"HW-GPT-Bench: Hardware-Aware Architecture Benchmark for Language Models","date":"2024-05-16","arxiv_id":"2405.10299","n_code_links":2,"syntology":null},{"paper":null,"slug":"optimization-techniques-for-sentiment","title":"Optimization Techniques for Sentiment Analysis Based on LLM (GPT-3)","date":"2024-05-16","arxiv_id":"2405.09770","n_code_links":0,"syntology":null},{"paper":null,"slug":"matching-domain-experts-by-training-from","title":"Matching domain experts by training from scratch on domain knowledge","date":"2024-05-15","arxiv_id":"2405.09395","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-scaling-laws-understanding-transformer","title":"Beyond Scaling Laws: Understanding Transformer Performance with Associative Memory","date":"2024-05-14","arxiv_id":"2405.08707","n_code_links":0,"syntology":null},{"paper":null,"slug":"challenges-in-deploying-long-context","title":"Challenges in Deploying Long-Context Transformers: A Theoretical Peak Performance Analysis","date":"2024-05-14","arxiv_id":"2405.08944","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-3-5-for-grammatical-error-correction","slug":"gpt-3-5-for-grammatical-error-correction","title":"GPT-3.5 for Grammatical Error Correction","date":"2024-05-14","arxiv_id":"2405.08469","n_code_links":0,"syntology":null},{"paper":null,"slug":"refinement-of-an-epilepsy-dictionary-through","title":"Refinement of an Epilepsy Dictionary through Human Annotation of Health-related posts on Instagram","date":"2024-05-14","arxiv_id":"2405.08784","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-principled-evaluations-of-sparse","title":"Towards Principled Evaluations of Sparse Autoencoders for Interpretability and Control","date":"2024-05-14","arxiv_id":"2405.08366","n_code_links":0,"syntology":null},{"paper":null,"slug":"when-large-language-models-meet-optical","title":"When Large Language Models Meet Optical Networks: Paving the Way for Automation","date":"2024-05-14","arxiv_id":"2405.17441","n_code_links":0,"syntology":null},{"paper":"/paper/can-language-models-explain-their-own","slug":"can-language-models-explain-their-own","title":"Can Language Models Explain Their Own Classification Behavior?","date":"2024-05-13","arxiv_id":"2405.07436","n_code_links":1,"syntology":null},{"paper":"/paper/coding-historical-causes-of-death-data-with","slug":"coding-historical-causes-of-death-data-with","title":"Coding historical causes of death data with Large Language Models","date":"2024-05-13","arxiv_id":"2405.07560","n_code_links":1,"syntology":null},{"paper":"/paper/freeva-offline-mllm-as-training-free-video","slug":"freeva-offline-mllm-as-training-free-video","title":"FreeVA: Offline MLLM as Training-Free Video Assistant","date":"2024-05-13","arxiv_id":"2405.07798","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":4,"n_instrument":4,"unverified":1,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["whwu95/freeva"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/macbehaviour-an-r-package-for-behavioural","slug":"macbehaviour-an-r-package-for-behavioural","title":"MacBehaviour: An R package for behavioural experimentation on large language models","date":"2024-05-13","arxiv_id":"2405.07495","n_code_links":1,"syntology":null},{"paper":null,"slug":"many-shot-regurgitation-msr-prompting","title":"Many-Shot Regurgitation (MSR) Prompting","date":"2024-05-13","arxiv_id":"2405.08134","n_code_links":0,"syntology":null},{"paper":null,"slug":"open-vocabulary-auditory-neural-decoding","title":"Open-vocabulary Auditory Neural Decoding Using fMRI-prompted LLM","date":"2024-05-13","arxiv_id":"2405.07840","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-reward-for-robot-skills-using-large","title":"Learning Reward for Robot Skills Using Large Language Models via Self-Alignment","date":"2024-05-12","arxiv_id":"2405.07162","n_code_links":0,"syntology":null},{"paper":"/paper/limited-ability-of-llms-to-simulate-human","slug":"limited-ability-of-llms-to-simulate-human","title":"Limited Ability of LLMs to Simulate Human Psychological Behaviours: a Psychometric Analysis","date":"2024-05-12","arxiv_id":"2405.07248","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nikbpetrov/llms-simulate-humans"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/quite-good-but-not-enough-nationality-bias-in","slug":"quite-good-but-not-enough-nationality-bias-in","title":"Quite Good, but Not Enough: Nationality Bias in Large Language Models -- A Case Study of ChatGPT","date":"2024-05-11","arxiv_id":"2405.06996","n_code_links":1,"syntology":null},{"paper":null,"slug":"retrieval-enhanced-zero-shot-video-captioning","title":"RETTA: Retrieval-Enhanced Test-Time Adaptation for Zero-Shot Video Captioning","date":"2024-05-11","arxiv_id":"2405.07046","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-assessment-of-model-on-model-deception","title":"An Assessment of Model-On-Model Deception","date":"2024-05-10","arxiv_id":"2405.12999","n_code_links":0,"syntology":null},{"paper":null,"slug":"chatgptest-opportunities-and-cautionary-tales","title":"ChatGPTest: opportunities and cautionary tales of utilizing AI for questionnaire pretesting","date":"2024-05-10","arxiv_id":"2405.06329","n_code_links":0,"syntology":null},{"paper":"/paper/a-mixture-of-experts-approach-to-3d-human","slug":"a-mixture-of-experts-approach-to-3d-human","title":"A Mixture of Experts Approach to 3D Human Motion Prediction","date":"2024-05-09","arxiv_id":"2405.06088","n_code_links":1,"syntology":null}],"record_sha256":"2f236c5a2c4abf1caa97ecb961837715d3faeb9cdbfe58abf0a911c34657099c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}