{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/cosine-annealing/papers/10","list_of":"/method/cosine-annealing","method":"Cosine Annealing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":10,"pages_in_order":40,"rows_per_page":100,"rows":[901,1000],"of":3965,"counts":{"archive_papers_tagged":3965,"with_a_code_link":1734,"where_syntology_ran_a_sample":627,"not_listed_spam_title":0,"listed":3965,"listed_where_code_ran":627,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":513,"every_run_a_failure_of_syntologys_instrument":114,"listed_with_a_run_with_no_instrument_failure":513,"listed_every_run_a_failure_of_syntologys_instrument":114,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/cosine-annealing","prev":"/method/cosine-annealing/papers/9","next":"/method/cosine-annealing/papers/11","papers":[{"paper":"/paper/evaluating-large-language-models-for-anxiety","slug":"evaluating-large-language-models-for-anxiety","title":"Evaluating Large Language Models for Anxiety and Depression Classification using Counseling and Psychotherapy Transcripts","date":"2024-07-18","arxiv_id":"2407.13228","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-as-reliable-knowledge","title":"How Reliable are LLMs as Knowledge Bases? Re-thinking Facutality and Consistency","date":"2024-07-18","arxiv_id":"2407.13578","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-from-mistakes-prompting-for","title":"Learning-From-Mistakes Prompting for Indigenous Language Translation","date":"2024-07-18","arxiv_id":"2407.13343","n_code_links":0,"syntology":null},{"paper":null,"slug":"pragyan-connecting-the-dots-in-tweets","title":"PRAGyan -- Connecting the Dots in Tweets","date":"2024-07-18","arxiv_id":"2407.13909","n_code_links":0,"syntology":null},{"paper":"/paper/werewolf-arena-a-case-study-in-llm-evaluation","slug":"werewolf-arena-a-case-study-in-llm-evaluation","title":"Werewolf Arena: A Case Study in LLM Evaluation via Social Deduction","date":"2024-07-18","arxiv_id":"2407.13943","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google/werewolf_arena"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"beyond-binary-multiclass-paraphasia-detection","title":"Beyond Binary: Multiclass Paraphasia Detection with Generative Pretrained Transformers and End-to-End Models","date":"2024-07-16","arxiv_id":"2407.11345","n_code_links":0,"syntology":null},{"paper":null,"slug":"chatbcg-can-ai-read-your-slide-deck","title":"ChatBCG: Can AI Read Your Slide Deck?","date":"2024-07-16","arxiv_id":"2407.12875","n_code_links":0,"syntology":null},{"paper":"/paper/does-refusal-training-in-llms-generalize-to","slug":"does-refusal-training-in-llms-generalize-to","title":"Does Refusal Training in LLMs Generalize to the Past Tense?","date":"2024-07-16","arxiv_id":"2407.11969","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tml-epfl/llm-past-tense"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gpt-assisted-annotation-of-rhetorical-and","title":"GPT Assisted Annotation of Rhetorical and Linguistic Features for Interpretable Propaganda Technique Detection in News Text","date":"2024-07-16","arxiv_id":"2407.11827","n_code_links":0,"syntology":null},{"paper":"/paper/lami-detr-open-vocabulary-detection-with","slug":"lami-detr-open-vocabulary-detection-with","title":"LaMI-DETR: Open-Vocabulary Detection with Language Model Instruction","date":"2024-07-16","arxiv_id":"2407.11335","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["eternaldolphin/lami-detr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-language-models-as-misleading","title":"Large Language Models as Misleading Assistants in Conversation","date":"2024-07-16","arxiv_id":"2407.11789","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-visual-language-models-are-also-good","title":"Large Visual-Language Models Are Also Good Classifiers: A Study of In-Context Multimodal Fake News Detection","date":"2024-07-16","arxiv_id":"2407.12879","n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-bias-in-political-sample","title":"Representation Bias in Political Sample Simulations with Large Language Models","date":"2024-07-16","arxiv_id":"2407.11409","n_code_links":0,"syntology":null},{"paper":null,"slug":"review-feedback-reason-refer-a-novel","title":"ReFeR: Improving Evaluation and Reasoning through Hierarchy of Models","date":"2024-07-16","arxiv_id":"2407.12877","n_code_links":0,"syntology":null},{"paper":"/paper/trust-no-bot-discovering-personal-disclosures","slug":"trust-no-bot-discovering-personal-disclosures","title":"Trust No Bot: Discovering Personal Disclosures in Human-LLM Conversations in the Wild","date":"2024-07-16","arxiv_id":"2407.11438","n_code_links":1,"syntology":null},{"paper":null,"slug":"empowering-llms-for-verilog-generation","title":"CodeV: Empowering LLMs with HDL Generation through Multi-Level Summarization","date":"2024-07-15","arxiv_id":"2407.10424","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-llm-respondents-for-item","title":"Leveraging LLM-Respondents for Item Evaluation: a Psychometric Analysis","date":"2024-07-15","arxiv_id":"2407.10899","n_code_links":0,"syntology":null},{"paper":null,"slug":"making-new-connections-llms-as-puzzle","title":"Making New Connections: LLMs as Puzzle Generators for The New York Times' Connections Word Game","date":"2024-07-15","arxiv_id":"2407.11240","n_code_links":0,"syntology":null},{"paper":null,"slug":"mechanistic-interpretability-of-large","title":"Mechanistic interpretability of large language models with applications to the financial services industry","date":"2024-07-15","arxiv_id":"2407.11215","n_code_links":0,"syntology":null},{"paper":"/paper/metallm-a-high-performant-and-cost-efficient","slug":"metallm-a-high-performant-and-cost-efficient","title":"MetaLLM: A High-performant and Cost-efficient Dynamic Framework for Wrapping LLMs","date":"2024-07-15","arxiv_id":"2407.10834","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mail-research/metallm-wrapper"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/think-on-graph-2-0-deep-and-interpretable","slug":"think-on-graph-2-0-deep-and-interpretable","title":"Think-on-Graph 2.0: Deep and Faithful Large Language Model Reasoning with Knowledge-guided Retrieval Augmented Generation","date":"2024-07-15","arxiv_id":"2407.10805","n_code_links":1,"syntology":{"ran":13,"of":16,"n_ran_checked":13,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["idea-finai/tog-2"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"curriculum-learning-for-small-code-language","title":"Curriculum Learning for Small Code Language Models","date":"2024-07-14","arxiv_id":"2407.10194","n_code_links":0,"syntology":null},{"paper":null,"slug":"document-level-clinical-entity-and-relation","title":"Document-level Clinical Entity and Relation Extraction via Knowledge Base-Guided Generation","date":"2024-07-13","arxiv_id":"2407.10021","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-in-store-customer-journeys-from","title":"Generating In-store Customer Journeys from Scratch with GPT Architectures","date":"2024-07-13","arxiv_id":"2407.11081","n_code_links":0,"syntology":null},{"paper":"/paper/astprompter-weakly-supervised-automated","slug":"astprompter-weakly-supervised-automated","title":"ASTPrompter: Weakly Supervised Automated Language Model Red-Teaming to Identify Low-Perplexity Toxic Prompts","date":"2024-07-12","arxiv_id":"2407.09447","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-evolving-gpt-a-lifelong-autonomous","title":"Self-Evolving GPT: A Lifelong Autonomous Experiential Learner","date":"2024-07-12","arxiv_id":"2407.08937","n_code_links":0,"syntology":null},{"paper":"/paper/show-don-t-tell-evaluating-large-language","slug":"show-don-t-tell-evaluating-large-language","title":"Show, Don't Tell: Evaluating Large Language Models Beyond Textual Understanding with ChildPlay","date":"2024-07-12","arxiv_id":"2407.11068","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-two-sides-of-the-coin-hallucination","title":"The Two Sides of the Coin: Hallucination Generation and Detection with LLMs as Evaluators for LLMs","date":"2024-07-12","arxiv_id":"2407.09152","n_code_links":0,"syntology":null},{"paper":null,"slug":"toward-automatic-group-membership-annotation","title":"Toward Automatic Group Membership Annotation for Group Fairness Evaluation","date":"2024-07-12","arxiv_id":"2407.08926","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-is-judged-more-human-than-humans-in","title":"GPT-4 is judged more human than humans in displaced and inverted Turing tests","date":"2024-07-11","arxiv_id":"2407.08853","n_code_links":0,"syntology":null},{"paper":null,"slug":"hdt-hierarchical-document-transformer","title":"HDT: Hierarchical Document Transformer","date":"2024-07-11","arxiv_id":"2407.08330","n_code_links":0,"syntology":null},{"paper":"/paper/llms-morphological-analyses-of-complex-fst","slug":"llms-morphological-analyses-of-complex-fst","title":"LLMs' morphological analyses of complex FST-generated Finnish words","date":"2024-07-11","arxiv_id":"2407.08269","n_code_links":1,"syntology":null},{"paper":"/paper/mavis-mathematical-visual-instruction-tuning","slug":"mavis-mathematical-visual-instruction-tuning","title":"MAVIS: Mathematical Visual Instruction Tuning with an Automatic Data Engine","date":"2024-07-11","arxiv_id":"2407.08739","n_code_links":3,"syntology":null},{"paper":null,"slug":"on-the-in-security-of-llm-app-stores","title":"On the (In)Security of LLM App Stores","date":"2024-07-11","arxiv_id":"2407.08422","n_code_links":0,"syntology":null},{"paper":null,"slug":"real-time-anomaly-detection-and-reactive","title":"Real-Time Anomaly Detection and Reactive Planning with Large Language Models","date":"2024-07-11","arxiv_id":"2407.08735","n_code_links":0,"syntology":null},{"paper":"/paper/vox-populi-vox-ai-using-language-models-to","slug":"vox-populi-vox-ai-using-language-models-to","title":"Vox Populi, Vox AI? Using Language Models to Estimate German Public Opinion","date":"2024-07-11","arxiv_id":"2407.08563","n_code_links":1,"syntology":null},{"paper":"/paper/fsponer-few-shot-prompt-optimization-for","slug":"fsponer-few-shot-prompt-optimization-for","title":"FsPONER: Few-shot Prompt Optimization for Named Entity Recognition in Domain-specific Scenarios","date":"2024-07-10","arxiv_id":"2407.08035","n_code_links":1,"syntology":null},{"paper":"/paper/kpopmt-translation-dataset-with-terminology","slug":"kpopmt-translation-dataset-with-terminology","title":"KpopMT: Translation Dataset with Terminology for Kpop Fandom","date":"2024-07-10","arxiv_id":"2407.07413","n_code_links":1,"syntology":null},{"paper":null,"slug":"mixsumm-topic-based-data-augmentation-using","title":"A Guide To Effectively Leveraging LLMs for Low-Resource Text Summarization: Data Augmentation and Semi-supervised Approaches","date":"2024-07-10","arxiv_id":"2407.07341","n_code_links":0,"syntology":null},{"paper":null,"slug":"multilingual-blending-llm-safety-alignment","title":"Multilingual Blending: LLM Safety Alignment Evaluation with Language Mixture","date":"2024-07-10","arxiv_id":"2407.07342","n_code_links":0,"syntology":null},{"paper":"/paper/rosa-random-subspace-adaptation-for-efficient","slug":"rosa-random-subspace-adaptation-for-efficient","title":"ROSA: Random Subspace Adaptation for Efficient Fine-Tuning","date":"2024-07-10","arxiv_id":"2407.07802","n_code_links":1,"syntology":null},{"paper":"/paper/ai-ai-bias-large-language-models-favor-their","slug":"ai-ai-bias-large-language-models-favor-their","title":"AI AI Bias: Large Language Models Favor Their Own Generated Content","date":"2024-07-09","arxiv_id":"2407.12856","n_code_links":1,"syntology":null},{"paper":"/paper/chatgpt-doesn-t-trust-chargers-fans-guardrail","slug":"chatgpt-doesn-t-trust-chargers-fans-guardrail","title":"ChatGPT Doesn't Trust Chargers Fans: Guardrail Sensitivity in Context","date":"2024-07-09","arxiv_id":"2407.06866","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["vli31/llm-guardrail-sensitivity"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"identification-of-emotions-on-twitter-during","title":"Identification of emotions on Twitter during the 2022 electoral process in Colombia","date":"2024-07-09","arxiv_id":"2407.07258","n_code_links":0,"syntology":null},{"paper":null,"slug":"measuring-sustainability-intention-of-esg","title":"Measuring Sustainability Intention of ESG Fund Disclosure using Few-Shot Learning","date":"2024-07-09","arxiv_id":"2407.06893","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompting-techniques-for-secure-code","title":"Prompting Techniques for Secure Code Generation: A Systematic Investigation","date":"2024-07-09","arxiv_id":"2407.07064","n_code_links":0,"syntology":null},{"paper":null,"slug":"raply-a-profanity-mitigated-rap-generator","title":"Raply: A profanity-mitigated rap generator","date":"2024-07-09","arxiv_id":"2407.06941","n_code_links":0,"syntology":null},{"paper":null,"slug":"solving-general-natural-language-description","title":"Solving General Natural-Language-Description Optimization Problems with Large Language Models","date":"2024-07-09","arxiv_id":"2407.07924","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-large-language-models-for-generating","title":"Using Large Language Models for Generating Smart Contracts for Health Insurance from Textual Policies","date":"2024-07-09","arxiv_id":"2407.07019","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-pretrained-large-language-model-with","title":"Using Pretrained Large Language Model with Prompt Engineering to Answer Biomedical Questions","date":"2024-07-09","arxiv_id":"2407.06779","n_code_links":0,"syntology":null},{"paper":"/paper/inversecoder-unleashing-the-power-of","slug":"inversecoder-unleashing-the-power-of","title":"InverseCoder: Self-improving Instruction-Tuned Code LLMs with Inverse-Instruct","date":"2024-07-08","arxiv_id":"2407.05700","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wyt2000/InverseCoder"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-understand-layouts","slug":"large-language-models-understand-layouts","title":"Large Language Models Understand Layout","date":"2024-07-08","arxiv_id":"2407.05750","n_code_links":1,"syntology":null},{"paper":null,"slug":"potential-of-multimodal-large-language-models","title":"Potential of Multimodal Large Language Models for Data Mining of Medical Images and Free-text Reports","date":"2024-07-08","arxiv_id":"2407.05758","n_code_links":0,"syntology":null},{"paper":null,"slug":"surprising-gender-biases-in-gpt","title":"Surprising gender biases in GPT","date":"2024-07-08","arxiv_id":"2407.06003","n_code_links":0,"syntology":null},{"paper":"/paper/shine-saliency-aware-hierarchical-negative","slug":"shine-saliency-aware-hierarchical-negative","title":"SHINE: Saliency-aware HIerarchical NEgative Ranking for Compositional Temporal Grounding","date":"2024-07-06","arxiv_id":"2407.05118","n_code_links":1,"syntology":{"ran":11,"of":13,"n_ran_checked":10,"n_instrument":1,"unverified":2,"pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zxccade/shine"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"are-large-language-models-strategic-decision","title":"Are Large Language Models Strategic Decision Makers? A Study of Performance and Bias in Two-Player Non-Zero-Sum Games","date":"2024-07-05","arxiv_id":"2407.04467","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-vs-retro-exploring-the-intersection-of","title":"GPT vs RETRO: Exploring the Intersection of Retrieval and Parameter-Efficient Fine-Tuning","date":"2024-07-05","arxiv_id":"2407.04528","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-data-to-commonsense-reasoning-the-use-of","title":"From Data to Commonsense Reasoning: The Use of Large Language Models for Explainable AI","date":"2024-07-04","arxiv_id":"2407.03778","n_code_links":0,"syntology":null},{"paper":null,"slug":"nutribench-a-dataset-for-evaluating-large","title":"NutriBench: A Dataset for Evaluating Large Language Models on Nutrition Estimation from Meal Descriptions","date":"2024-07-04","arxiv_id":"2407.12843","n_code_links":0,"syntology":null},{"paper":"/paper/planning-with-large-language-models-for","slug":"planning-with-large-language-models-for","title":"Controllable Conversations: Planning-Based Dialogue Agent with Large Language Models","date":"2024-07-04","arxiv_id":"2407.03884","n_code_links":1,"syntology":null},{"paper":null,"slug":"question-analysis-prompting-improves-llm","title":"Question-Analysis Prompting Improves LLM Performance in Reasoning Tasks","date":"2024-07-04","arxiv_id":"2407.03624","n_code_links":0,"syntology":null},{"paper":null,"slug":"slice-100k-a-multimodal-dataset-for-extrusion","title":"Slice-100K: A Multimodal Dataset for Extrusion-based 3D Printing","date":"2024-07-04","arxiv_id":"2407.04180","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-automating-text-annotation-a-case","title":"Towards Automating Text Annotation: A Case Study on Semantic Proximity Annotation using GPT-4","date":"2024-07-04","arxiv_id":"2407.04130","n_code_links":0,"syntology":null},{"paper":null,"slug":"agentinstruct-toward-generative-teaching-with","title":"AgentInstruct: Toward Generative Teaching with Agentic Flows","date":"2024-07-03","arxiv_id":"2407.03502","n_code_links":0,"syntology":null},{"paper":"/paper/automatic-gradient-descent-with-generalized","slug":"automatic-gradient-descent-with-generalized","title":"Gradient descent with generalized Newton's method","date":"2024-07-03","arxiv_id":"2407.02772","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shiyunxu/autogen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-llm-abilities-in-idiomatic","title":"Improving LLM Abilities in Idiomatic Translation","date":"2024-07-03","arxiv_id":"2407.03518","n_code_links":0,"syntology":null},{"paper":null,"slug":"obfuscatune-obfuscated-offsite-fine-tuning","title":"ObfuscaTune: Obfuscated Offsite Fine-tuning and Inference of Proprietary LLMs on Private Datasets","date":"2024-07-03","arxiv_id":"2407.02960","n_code_links":0,"syntology":null},{"paper":null,"slug":"ospc-artificial-vlm-features-for-hateful-meme","title":"OSPC: Artificial VLM Features for Hateful Meme Detection","date":"2024-07-03","arxiv_id":"2407.12836","n_code_links":0,"syntology":null},{"paper":null,"slug":"regurgitative-training-the-value-of-real-data","title":"Regurgitative Training: The Value of Real Data in Training Large Language Models","date":"2024-07-03","arxiv_id":"2407.12835","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-the-code-clone-detection-capability","title":"Assessing the Code Clone Detection Capability of Large Language Models","date":"2024-07-02","arxiv_id":"2407.02402","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-numeric-awards-in-context-dueling","title":"Beyond Numeric Awards: In-Context Dueling Bandits with LLM Agents","date":"2024-07-02","arxiv_id":"2407.01887","n_code_links":0,"syntology":null},{"paper":"/paper/gptcast-a-weather-language-model-for","slug":"gptcast-a-weather-language-model-for","title":"GPTCast: a weather language model for precipitation nowcasting","date":"2024-07-02","arxiv_id":"2407.02089","n_code_links":1,"syntology":null},{"paper":null,"slug":"grasp-a-grid-based-benchmark-for-evaluating","title":"GRASP: A Grid-Based Benchmark for Evaluating Commonsense Spatial Reasoning","date":"2024-07-02","arxiv_id":"2407.01892","n_code_links":0,"syntology":null},{"paper":"/paper/integrate-the-essence-and-eliminate-the-dross","slug":"integrate-the-essence-and-eliminate-the-dross","title":"Integrate the Essence and Eliminate the Dross: Fine-Grained Self-Consistency for Free-Form Language Generation","date":"2024-07-02","arxiv_id":"2407.02056","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["WangXinglin/FSC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/multilingual-trolley-problems-for-language","slug":"multilingual-trolley-problems-for-language","title":"Language Model Alignment in Multilingual Trolley Problems","date":"2024-07-02","arxiv_id":"2407.02273","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["causalNLP/moralmachine","causalnlp/multitp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/sop-unlock-the-power-of-social-facilitation","slug":"sop-unlock-the-power-of-social-facilitation","title":"SeqAR: Jailbreak LLMs with Sequential Auto-Generated Characters","date":"2024-07-02","arxiv_id":"2407.01902","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yang-yan-yang-yan/sop"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/increasing-model-capacity-for-free-a-simple","slug":"increasing-model-capacity-for-free-a-simple","title":"Increasing Model Capacity for Free: A Simple Strategy for Parameter Efficient Fine-tuning","date":"2024-07-01","arxiv_id":"2407.01320","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":6,"n_instrument":1,"unverified":3,"pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["lins-lab/capaboost"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"predicting-dc-link-capacitor-current-ripple","title":"Predicting DC-Link Capacitor Current Ripple in AC-DC Rectifier Circuits Using Fine-Tuned Large Language Models","date":"2024-07-01","arxiv_id":"2407.01724","n_code_links":0,"syntology":null},{"paper":"/paper/parm-efficient-training-of-large-sparsely","slug":"parm-efficient-training-of-large-sparsely","title":"Parm: Efficient Training of Large Sparsely-Activated Models with Dedicated Schedules","date":"2024-06-30","arxiv_id":"2407.00599","n_code_links":1,"syntology":null},{"paper":null,"slug":"applying-rlaif-for-code-generation-with-api","title":"Applying RLAIF for Code Generation with API-usage in Lightweight LLMs","date":"2024-06-28","arxiv_id":"2406.20060","n_code_links":0,"syntology":null},{"paper":null,"slug":"fred-flexible-reduction-distribution","title":"FRED: Flexible REduction-Distribution Interconnect and Communication Implementation for Wafer-Scale Distributed Training of DNN Models","date":"2024-06-28","arxiv_id":"2406.19580","n_code_links":0,"syntology":null},{"paper":"/paper/machine-learning-predictors-for-min-entropy","slug":"machine-learning-predictors-for-min-entropy","title":"Machine Learning Predictors for Min-Entropy Estimation","date":"2024-06-28","arxiv_id":"2406.19983","n_code_links":1,"syntology":null},{"paper":null,"slug":"scalebio-scalable-bilevel-optimization-for","title":"ScaleBiO: Scalable Bilevel Optimization for LLM Data Reweighting","date":"2024-06-28","arxiv_id":"2406.19976","n_code_links":0,"syntology":null},{"paper":"/paper/shortcutsbench-a-large-scale-real-world","slug":"shortcutsbench-a-large-scale-real-world","title":"ShortcutsBench: A Large-Scale Real-world Benchmark for API-based Agents","date":"2024-06-28","arxiv_id":"2407.00132","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":4,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["eachsheep/shortcutsbench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/fibottention-inceptive-visual-representation","slug":"fibottention-inceptive-visual-representation","title":"Fibottention: Inceptive Visual Representation Learning with Diverse Attention Across Heads","date":"2024-06-27","arxiv_id":"2406.19391","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-tuned-network-relies-on-generic","title":"Fine-tuned network relies on generic representation to solve unseen cognitive task","date":"2024-06-27","arxiv_id":"2406.18926","n_code_links":0,"syntology":null},{"paper":"/paper/from-artificial-needles-to-real-haystacks","slug":"from-artificial-needles-to-real-haystacks","title":"From Artificial Needles to Real Haystacks: Improving Retrieval Capabilities in LLMs by Finetuning on Synthetic Data","date":"2024-06-27","arxiv_id":"2406.19292","n_code_links":1,"syntology":null},{"paper":null,"slug":"granite-function-calling-model-introducing","title":"Granite-Function Calling Model: Introducing Function Calling Abilities via Multi-task Learning of Granular Tasks","date":"2024-06-27","arxiv_id":"2407.00121","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-model-arena-for-cross-lingual-sentiment","title":"The Model Arena for Cross-lingual Sentiment Analysis: A Comparative Study in the Era of Large Language Models","date":"2024-06-27","arxiv_id":"2406.19358","n_code_links":0,"syntology":null},{"paper":null,"slug":"apigen-automated-pipeline-for-generating","title":"APIGen: Automated Pipeline for Generating Verifiable and Diverse Function-Calling Datasets","date":"2024-06-26","arxiv_id":"2406.18518","n_code_links":0,"syntology":null},{"paper":"/paper/factfinders-at-checkthat-2024-refining-check","slug":"factfinders-at-checkthat-2024-refining-check","title":"FactFinders at CheckThat! 2024: Refining Check-worthy Statement Detection with LLMs through Data Pruning","date":"2024-06-26","arxiv_id":"2406.18297","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-entity-recognition-using-ensembles","title":"Improving Entity Recognition Using Ensembles of Deep Learning and Fine-tuned Large Language Models: A Case Study on Adverse Event Extraction from Multiple Sources","date":"2024-06-26","arxiv_id":"2406.18049","n_code_links":0,"syntology":null},{"paper":"/paper/mathodyssey-benchmarking-mathematical-problem","slug":"mathodyssey-benchmarking-mathematical-problem","title":"MathOdyssey: Benchmarking Mathematical Problem-Solving Skills in Large Language Models Using Odyssey Math Data","date":"2024-06-26","arxiv_id":"2406.18321","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"bert-neural-information-retrieval-boolean","title":"SetBERT: Enhancing Retrieval Performance for Boolean Logic and Set Operation Queries","date":"2024-06-25","arxiv_id":"2406.17282","n_code_links":0,"syntology":null},{"paper":"/paper/interpreting-attention-layer-outputs-with","slug":"interpreting-attention-layer-outputs-with","title":"Interpreting Attention Layer Outputs with Sparse Autoencoders","date":"2024-06-25","arxiv_id":"2406.17759","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["ckkissane/attention-output-saes"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"what-do-the-circuits-mean-a-knowledge-edit","title":"Understanding Language Model Circuits through Knowledge Editing","date":"2024-06-25","arxiv_id":"2406.17241","n_code_links":0,"syntology":null},{"paper":"/paper/dreambench-a-human-aligned-benchmark-for","slug":"dreambench-a-human-aligned-benchmark-for","title":"DreamBench++: A Human-Aligned Benchmark for Personalized Image Generation","date":"2024-06-24","arxiv_id":"2406.16855","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yuangpeng/dreambench_plus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluation-of-instruction-following-ability","title":"Evaluation of Instruction-Following Ability for Large Language Models on Story-Ending Generation","date":"2024-06-24","arxiv_id":"2406.16356","n_code_links":0,"syntology":null},{"paper":"/paper/finding-transformer-circuits-with-edge","slug":"finding-transformer-circuits-with-edge","title":"Finding Transformer Circuits with Edge Pruning","date":"2024-06-24","arxiv_id":"2406.16778","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":3,"n_instrument":5,"unverified":0,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["princeton-nlp/edge-pruning"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"modeling-a-novel-dataset-for-testing","title":"modeLing: A Novel Dataset for Testing Linguistic Reasoning in Language Models","date":"2024-06-24","arxiv_id":"2406.17038","n_code_links":0,"syntology":null}],"record_sha256":"3c4fae76caabe6474af2eab409b9deaa806f9d5919cf18e7f2f8511275b4595e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}