{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/discriminative-fine-tuning/papers/3","list_of":"/method/discriminative-fine-tuning","method":"Discriminative Fine-Tuning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":3,"pages_in_order":20,"rows_per_page":100,"rows":[201,300],"of":1990,"counts":{"archive_papers_tagged":1990,"with_a_code_link":794,"where_syntology_ran_a_sample":271,"not_listed_spam_title":0,"listed":1990,"listed_where_code_ran":271,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":223,"every_run_a_failure_of_syntologys_instrument":48,"listed_with_a_run_with_no_instrument_failure":223,"listed_every_run_a_failure_of_syntologys_instrument":48,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/discriminative-fine-tuning","prev":"/method/discriminative-fine-tuning/papers/2","next":"/method/discriminative-fine-tuning/papers/4","papers":[{"paper":null,"slug":"ynote-a-novel-music-notation-for-fine-tuning","title":"YNote: A Novel Music Notation for Fine-Tuning LLMs in Music Generation","date":"2025-02-12","arxiv_id":"2502.10467","n_code_links":0,"syntology":null},{"paper":"/paper/automated-capability-discovery-via-model-self","slug":"automated-capability-discovery-via-model-self","title":"Automated Capability Discovery via Model Self-Exploration","date":"2025-02-11","arxiv_id":"2502.07577","n_code_links":2,"syntology":null},{"paper":null,"slug":"debatebench-a-challenging-long-context","title":"DebateBench: A Challenging Long Context Reasoning Benchmark For Large Language Models","date":"2025-02-10","arxiv_id":"2502.06279","n_code_links":0,"syntology":null},{"paper":null,"slug":"find-central-dogma-again","title":"Find Central Dogma Again: Leveraging Multilingual Transfer in Large Language Models","date":"2025-02-10","arxiv_id":"2502.06253","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-in-software-security-a","title":"LLMs in Software Security: A Survey of Vulnerability Detection Techniques and Insights","date":"2025-02-10","arxiv_id":"2502.07049","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-gpt-4o-efficiency-for-detecting","title":"Leveraging GPT-4o Efficiency for Detecting Rework Anomaly in Business Processes","date":"2025-02-10","arxiv_id":"2502.06918","n_code_links":0,"syntology":null},{"paper":null,"slug":"emergence-of-episodic-memory-in-transformers","title":"Emergence of Episodic Memory in Transformers: Characterizing Changes in Temporal Structure of Attention Scores During Training","date":"2025-02-09","arxiv_id":"2502.06902","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-for-in-file","title":"Large Language Models for In-File Vulnerability Localization Can Be \"Lost in the End\"","date":"2025-02-09","arxiv_id":"2502.06898","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaffoldgpt-a-scaffold-based-large-language","title":"ScaffoldGPT: A Scaffold-based GPT Model for Drug Optimization","date":"2025-02-09","arxiv_id":"2502.06891","n_code_links":0,"syntology":null},{"paper":"/paper/the-odyssey-of-the-fittest-can-agents-survive","slug":"the-odyssey-of-the-fittest-can-agents-survive","title":"The Odyssey of the Fittest: Can Agents Survive and Still Be Good?","date":"2025-02-08","arxiv_id":"2502.05442","n_code_links":1,"syntology":null},{"paper":null,"slug":"detection-of-llm-generated-java-code-using","title":"Detection of LLM-Generated Java Code Using Discretized Nested Bigrams","date":"2025-02-07","arxiv_id":"2502.15740","n_code_links":0,"syntology":null},{"paper":null,"slug":"eap-gp-mitigating-saturation-effect-in","title":"EAP-GP: Mitigating Saturation Effect in Gradient-based Automated Circuit Identification","date":"2025-02-07","arxiv_id":"2502.06852","n_code_links":0,"syntology":null},{"paper":"/paper/llasa-scaling-train-time-and-inference-time","slug":"llasa-scaling-train-time-and-inference-time","title":"Llasa: Scaling Train-Time and Inference-Time Compute for Llama-based Speech Synthesis","date":"2025-02-06","arxiv_id":"2502.04128","n_code_links":1,"syntology":null},{"paper":null,"slug":"aligning-human-and-machine-attention-for","title":"Aligning Human and Machine Attention for Enhanced Supervised Learning","date":"2025-02-04","arxiv_id":"2502.06811","n_code_links":0,"syntology":null},{"paper":null,"slug":"conversation-ai-dialog-for-medicare-powered","title":"Conversation AI Dialog for Medicare powered by Finetuning and Retrieval Augmented Generation","date":"2025-02-04","arxiv_id":"2502.02249","n_code_links":0,"syntology":null},{"paper":"/paper/harmonic-loss-trains-interpretable-ai-models","slug":"harmonic-loss-trains-interpretable-ai-models","title":"Harmonic Loss Trains Interpretable AI Models","date":"2025-02-03","arxiv_id":"2502.01628","n_code_links":1,"syntology":null},{"paper":"/paper/learnable-polynomial-trigonometric-and","slug":"learnable-polynomial-trigonometric-and","title":"Polynomial, trigonometric, and tropical activations","date":"2025-02-03","arxiv_id":"2502.01247","n_code_links":1,"syntology":null},{"paper":null,"slug":"libra-measuring-bias-of-large-language-model","title":"LIBRA: Measuring Bias of Large Language Model from a Local Context","date":"2025-02-02","arxiv_id":"2502.01679","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-solvers-for-discrete-diffusion-models","title":"Fast Solvers for Discrete Diffusion Models: Theory and Applications of High-Order Algorithms","date":"2025-02-01","arxiv_id":"2502.00234","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-ai-solve-the-peer-review-crisis-a-large","title":"Can AI Solve the Peer Review Crisis? A Large Scale Cross Model Experiment of LLMs' Performance and Biases in Evaluating over 1000 Economics Papers","date":"2025-01-31","arxiv_id":"2502.00070","n_code_links":0,"syntology":null},{"paper":null,"slug":"alphaadam-asynchronous-masked-optimization","title":"AlphaAdam:Asynchronous Masked Optimization with Dynamic Alpha for Selective Updates","date":"2025-01-30","arxiv_id":"2501.18094","n_code_links":0,"syntology":null},{"paper":null,"slug":"economic-rationality-under-specialization","title":"Economic Rationality under Specialization: Evidence of Decision Bias in AI Agents","date":"2025-01-30","arxiv_id":"2501.18190","n_code_links":0,"syntology":null},{"paper":null,"slug":"structure-development-in-list-sorting","title":"Structure Development in List-Sorting Transformers","date":"2025-01-30","arxiv_id":"2501.18666","n_code_links":0,"syntology":null},{"paper":"/paper/wildchat-50m-a-deep-dive-into-the-role-of","slug":"wildchat-50m-a-deep-dive-into-the-role-of","title":"WILDCHAT-50M: A Deep Dive Into the Role of Synthetic Data in Post-Training","date":"2025-01-30","arxiv_id":"2501.18511","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["penfever/wildchat-50m"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"open-source-retrieval-augmented-generation","title":"Open-Source Retrieval Augmented Generation Framework for Retrieving Accurate Medication Insights from Formularies for African Healthcare Workers","date":"2025-01-28","arxiv_id":"2502.15722","n_code_links":0,"syntology":null},{"paper":null,"slug":"kernels-of-selfhood-gpt-4o-shows-humanlike","title":"Kernels of Selfhood: GPT-4o shows humanlike patterns of cognitive consistency moderated by free choice","date":"2025-01-27","arxiv_id":"2502.07088","n_code_links":0,"syntology":null},{"paper":null,"slug":"weight-based-analysis-of-detokenization-in","title":"Weight-based Analysis of Detokenization in Language Models: Understanding the First Stage of Inference Without Inference","date":"2025-01-27","arxiv_id":"2501.15754","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-attempt-to-unraveling-token-prediction","title":"An Attempt to Unraveling Token Prediction Refinement and Identifying Essential Layers of Large Language Models","date":"2025-01-25","arxiv_id":"2501.15054","n_code_links":0,"syntology":null},{"paper":null,"slug":"darkmind-latent-chain-of-thought-backdoor-in","title":"DarkMind: Latent Chain-of-Thought Backdoor in Customized LLMs","date":"2025-01-24","arxiv_id":"2501.18617","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-are-vulnerable-to-malicious-prompts","title":"LLMs are Vulnerable to Malicious Prompts Disguised as Scientific Language","date":"2025-01-23","arxiv_id":"2501.14073","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-gpt-s-ability-as-a-judge-in-music","slug":"exploring-gpt-s-ability-as-a-judge-in-music","title":"Exploring GPT's Ability as a Judge in Music Understanding","date":"2025-01-22","arxiv_id":"2501.13261","n_code_links":1,"syntology":null},{"paper":"/paper/advancing-the-understanding-and-evaluation-of","slug":"advancing-the-understanding-and-evaluation-of","title":"Advancing the Understanding and Evaluation of AR-Generated Scenes: When Vision-Language Models Shine and Stumble","date":"2025-01-21","arxiv_id":"2501.13964","n_code_links":1,"syntology":null},{"paper":null,"slug":"focus-first-order-concentrated-updating","title":"FOCUS: First Order Concentrated Updating Scheme","date":"2025-01-21","arxiv_id":"2501.12243","n_code_links":0,"syntology":null},{"paper":null,"slug":"harnessing-generative-pre-trained-transformer","title":"Harnessing Generative Pre-Trained Transformer for Datacenter Packet Trace Generation","date":"2025-01-21","arxiv_id":"2501.12033","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-language-models-for-automated-chest-x","title":"Vision-Language Models for Automated Chest X-ray Interpretation: Leveraging ViT and GPT-2","date":"2025-01-21","arxiv_id":"2501.12356","n_code_links":0,"syntology":null},{"paper":null,"slug":"fsmoe-a-flexible-and-scalable-training-system","title":"FSMoE: A Flexible and Scalable Training System for Sparse Mixture-of-Experts Models","date":"2025-01-18","arxiv_id":"2501.10714","n_code_links":0,"syntology":null},{"paper":"/paper/confidence-estimation-for-error-detection-in","slug":"confidence-estimation-for-error-detection-in","title":"Confidence Estimation for Error Detection in Text-to-SQL Systems","date":"2025-01-16","arxiv_id":"2501.09527","n_code_links":1,"syntology":null},{"paper":null,"slug":"generative-ai-takes-a-statistics-exam-a","title":"Generative AI Takes a Statistics Exam: A Comparison of Performance between ChatGPT3.5, ChatGPT4, and ChatGPT4o-mini","date":"2025-01-15","arxiv_id":"2501.09171","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-energy-efficiency-and","title":"Investigating Energy Efficiency and Performance Trade-offs in LLM Inference Across Tasks and DVFS Settings","date":"2025-01-14","arxiv_id":"2501.08219","n_code_links":0,"syntology":null},{"paper":"/paper/finerweb-10bt-refining-web-data-with-llm","slug":"finerweb-10bt-refining-web-data-with-llm","title":"FinerWeb-10BT: Refining Web Data with LLM-Based Line-Level Filtering","date":"2025-01-13","arxiv_id":"2501.07314","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-as-a-monte-carlo-language-tree-a","title":"GPT as a Monte Carlo Language Tree: A Probabilistic Perspective","date":"2025-01-13","arxiv_id":"2501.07641","n_code_links":0,"syntology":null},{"paper":"/paper/how-gpt-learns-layer-by-layer","slug":"how-gpt-learns-layer-by-layer","title":"How GPT learns layer by layer","date":"2025-01-13","arxiv_id":"2501.07108","n_code_links":1,"syntology":null},{"paper":"/paper/assessing-instructor-ai-cooperation-for","slug":"assessing-instructor-ai-cooperation-for","title":"Assessing instructor-AI cooperation for grading essay-type questions in an introductory sociology course","date":"2025-01-11","arxiv_id":"2501.06461","n_code_links":1,"syntology":null},{"paper":"/paper/uav-vla-vision-language-action-system-for","slug":"uav-vla-vision-language-action-system-for","title":"UAV-VLA: Vision-Language-Action System for Large Scale Aerial Mission Generation","date":"2025-01-09","arxiv_id":"2501.05014","n_code_links":1,"syntology":null},{"paper":null,"slug":"integrating-llms-with-its-recent-advances","title":"Integrating LLMs with ITS: Recent Advances, Potentials, Challenges, and Future Directions","date":"2025-01-08","arxiv_id":"2501.04437","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-large-language-model-training-on","title":"Scaling Large Language Model Training on Frontier with Low-Bandwidth Partitioning","date":"2025-01-08","arxiv_id":"2501.04266","n_code_links":0,"syntology":null},{"paper":"/paper/finding-a-voice-evaluating-african-american","slug":"finding-a-voice-evaluating-african-american","title":"Finding A Voice: Evaluating African American Dialect Generation for Chatbot Technology","date":"2025-01-07","arxiv_id":"2501.03441","n_code_links":1,"syntology":null},{"paper":"/paper/decoding-fmri-data-into-captions-using-prefix","slug":"decoding-fmri-data-into-captions-using-prefix","title":"Decoding fMRI Data into Captions using Prefix Language Modeling","date":"2025-01-05","arxiv_id":"2501.02570","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-survey-on-large-language-models-with-some","title":"A Survey on Large Language Models with some Insights on their Capabilities and Limitations","date":"2025-01-03","arxiv_id":"2501.04040","n_code_links":0,"syntology":null},{"paper":null,"slug":"agentrefine-enhancing-agent-generalization","title":"AgentRefine: Enhancing Agent Generalization through Refinement Tuning","date":"2025-01-03","arxiv_id":"2501.01702","n_code_links":0,"syntology":null},{"paper":"/paper/crrg-clip-automatic-generation-of-chest","slug":"crrg-clip-automatic-generation-of-chest","title":"CRRG-CLIP: Automatic Generation of Chest Radiology Reports and Classification of Chest Radiographs","date":"2024-12-31","arxiv_id":"2501.01989","n_code_links":1,"syntology":null},{"paper":null,"slug":"why-are-positional-encodings-nonessential-for","title":"Why Are Positional Encodings Nonessential for Deep Autoregressive Transformers? Revisiting a Petroglyph","date":"2024-12-31","arxiv_id":"2501.00659","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparative-performance-of-advanced-nlp","title":"Comparative Performance of Advanced NLP Models and LLMs in Multilingual Geo-Entity Detection","date":"2024-12-29","arxiv_id":"2412.20414","n_code_links":0,"syntology":null},{"paper":"/paper/electra-and-gpt-4o-cost-effective-partners","slug":"electra-and-gpt-4o-cost-effective-partners","title":"ELECTRA and GPT-4o: Cost-Effective Partners for Sentiment Analysis","date":"2024-12-29","arxiv_id":"2501.00062","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-adversarial-robustness-of-language-models","title":"On Adversarial Robustness of Language Models in Transfer Learning","date":"2024-12-29","arxiv_id":"2501.00066","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-text-classification-methods-for","title":"Assessing Text Classification Methods for Cyberbullying Detection on Social Media Platforms","date":"2024-12-27","arxiv_id":"2412.19928","n_code_links":0,"syntology":null},{"paper":"/paper/drivingworld-constructingworld-model-for","slug":"drivingworld-constructingworld-model-for","title":"DrivingWorld: Constructing World Model for Autonomous Driving via Video GPT","date":"2024-12-27","arxiv_id":"2412.19505","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":8,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yvanyin/drivingworld"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/generative-pretrained-embedding-and","slug":"generative-pretrained-embedding-and","title":"Generative Pretrained Embedding and Hierarchical Irregular Time Series Representation for Daily Living Activity Recognition","date":"2024-12-27","arxiv_id":"2412.19732","n_code_links":1,"syntology":null},{"paper":null,"slug":"do-language-models-understand-the-cognitive","title":"Do Language Models Understand the Cognitive Tasks Given to Them? Investigations with the N-Back Paradigm","date":"2024-12-24","arxiv_id":"2412.18120","n_code_links":0,"syntology":null},{"paper":null,"slug":"bridging-auditory-perception-and-language","title":"Bridging Auditory Perception and Language Comprehension through MEG-Driven Encoding Models","date":"2024-12-22","arxiv_id":"2501.03246","n_code_links":0,"syntology":null},{"paper":"/paper/psychadapter-adapting-llm-transformers-to","slug":"psychadapter-adapting-llm-transformers-to","title":"PsychAdapter: Adapting LLM Transformers to Reflect Traits, Personality and Mental Health","date":"2024-12-22","arxiv_id":"2412.16882","n_code_links":1,"syntology":null},{"paper":null,"slug":"robustness-of-large-language-models-against","title":"Robustness of Large Language Models Against Adversarial Attacks","date":"2024-12-22","arxiv_id":"2412.17011","n_code_links":0,"syntology":null},{"paper":null,"slug":"tar3d-creating-high-quality-3d-assets-via","title":"TAR3D: Creating High-Quality 3D Assets via Next-Part Prediction","date":"2024-12-22","arxiv_id":"2412.16919","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-the-performance-of-large-language-4","title":"Evaluating the Performance of Large Language Models in Scientific Claim Detection and Classification","date":"2024-12-21","arxiv_id":"2412.16486","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-fim-code-completions-via-context","title":"Improving FIM Code Completions via Context & Curriculum Based Learning","date":"2024-12-21","arxiv_id":"2412.16589","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-robustness-through-dynamic","title":"Adversarial Robustness through Dynamic Ensemble Learning","date":"2024-12-20","arxiv_id":"2412.16254","n_code_links":0,"syntology":null},{"paper":"/paper/linguistic-features-extracted-by-gpt-4","slug":"linguistic-features-extracted-by-gpt-4","title":"Linguistic Features Extracted by GPT-4 Improve Alzheimer's Disease Detection based on Spontaneous Speech","date":"2024-12-20","arxiv_id":"2412.15772","n_code_links":1,"syntology":null},{"paper":null,"slug":"graph-convolutional-networks-named-entity","title":"Graph-Convolutional Networks: Named Entity Recognition and Large Language Model Embedding in Document Clustering","date":"2024-12-19","arxiv_id":"2412.14867","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-good-is-gpt-at-writing-political-speeches","title":"How good is GPT at writing political speeches for the White House?","date":"2024-12-19","arxiv_id":"2412.14617","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-as-mediators-can-they-diagnose-conflicts","title":"LLMs as mediators: Can they diagnose conflicts accurately?","date":"2024-12-19","arxiv_id":"2412.14675","n_code_links":0,"syntology":null},{"paper":null,"slug":"relational-programming-with-foundation-models","title":"Relational Programming with Foundation Models","date":"2024-12-19","arxiv_id":"2412.14515","n_code_links":0,"syntology":null},{"paper":"/paper/resofilter-rine-grained-synthetic-data","slug":"resofilter-rine-grained-synthetic-data","title":"ResoFilter: Fine-grained Synthetic Data Filtering for Large Language Models through Data-Parameter Resonance Analysis","date":"2024-12-19","arxiv_id":"2412.14809","n_code_links":1,"syntology":null},{"paper":"/paper/mix-ln-unleashing-the-power-of-deeper-layers","slug":"mix-ln-unleashing-the-power-of-deeper-layers","title":"Mix-LN: Unleashing the Power of Deeper Layers by Combining Pre-LN and Post-LN","date":"2024-12-18","arxiv_id":"2412.13795","n_code_links":1,"syntology":null},{"paper":null,"slug":"detecting-document-level-paraphrased-machine","title":"Detecting Document-level Paraphrased Machine Generated Content: Mimicking Human Writing Style and Involving Discourse Features","date":"2024-12-17","arxiv_id":"2412.12679","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-are-also-effective-embedding-models-an","title":"LLMs are Also Effective Embedding Models: An In-depth Overview","date":"2024-12-17","arxiv_id":"2412.12591","n_code_links":0,"syntology":null},{"paper":"/paper/causal-diffusion-transformers-for-generative","slug":"causal-diffusion-transformers-for-generative","title":"Causal Diffusion Transformers for Generative Modeling","date":"2024-12-16","arxiv_id":"2412.12095","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":6,"n_instrument":2,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["causalfusion/causalfusion"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/look-ahead-text-understanding-and-llm","slug":"look-ahead-text-understanding-and-llm","title":"Look Ahead Text Understanding and LLM Stitching","date":"2024-12-16","arxiv_id":"2412.17836","n_code_links":1,"syntology":null},{"paper":"/paper/no-more-adam-learning-rate-scaling-at","slug":"no-more-adam-learning-rate-scaling-at","title":"No More Adam: Learning Rate Scaling at Initialization is All You Need","date":"2024-12-16","arxiv_id":"2412.11768","n_code_links":1,"syntology":null},{"paper":null,"slug":"priority-aware-model-distributed-inference-at","title":"Priority-Aware Model-Distributed Inference at Edge Networks","date":"2024-12-16","arxiv_id":"2412.12371","n_code_links":0,"syntology":null},{"paper":"/paper/do-large-language-vision-models-understand-3d","slug":"do-large-language-vision-models-understand-3d","title":"Do large language vision models understand 3D shapes?","date":"2024-12-14","arxiv_id":"2412.10908","n_code_links":1,"syntology":null},{"paper":"/paper/does-multiple-choice-have-a-future-in-the-age","slug":"does-multiple-choice-have-a-future-in-the-age","title":"Does Multiple Choice Have a Future in the Age of Generative AI? A Posttest-only RCT","date":"2024-12-13","arxiv_id":"2412.10267","n_code_links":1,"syntology":null},{"paper":"/paper/causal-world-representation-in-the-gpt-model","slug":"causal-world-representation-in-the-gpt-model","title":"A Causal World Model Underlying Next Token Prediction: Exploring GPT in a Controlled Environment","date":"2024-12-10","arxiv_id":"2412.07446","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-2-through-the-lens-of-vector-symbolic","title":"GPT-2 Through the Lens of Vector Symbolic Architectures","date":"2024-12-10","arxiv_id":"2412.07947","n_code_links":0,"syntology":null},{"paper":"/paper/rag-based-question-answering-over","slug":"rag-based-question-answering-over","title":"RAG-based Question Answering over Heterogeneous Data and Text","date":"2024-12-10","arxiv_id":"2412.07420","n_code_links":0,"syntology":null},{"paper":"/paper/superficial-consciousness-hypothesis-for","slug":"superficial-consciousness-hypothesis-for","title":"Superficial Consciousness Hypothesis for Autoregressive Transformers","date":"2024-12-10","arxiv_id":"2412.07278","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-predictive-communication-with-brain","title":"Towards Predictive Communication with Brain-Computer Interfaces integrating Large Language Models","date":"2024-12-10","arxiv_id":"2412.07355","n_code_links":0,"syntology":null},{"paper":"/paper/batchtopk-sparse-autoencoders","slug":"batchtopk-sparse-autoencoders","title":"BatchTopK Sparse Autoencoders","date":"2024-12-09","arxiv_id":"2412.06410","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["bartbussmann/batchtopk"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-memorization-and-copyright","title":"Exploring Memorization and Copyright Violation in Frontier LLMs: A Study of the New York Times v. OpenAI 2023 Lawsuit","date":"2024-12-09","arxiv_id":"2412.06370","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-rosetta-paradox-domain-specific","title":"The Rosetta Paradox: Domain-Specific Performance Inversions in Large Language Models","date":"2024-12-09","arxiv_id":"2412.17821","n_code_links":0,"syntology":null},{"paper":"/paper/characterbox-evaluating-the-role-playing","slug":"characterbox-evaluating-the-role-playing","title":"CharacterBox: Evaluating the Role-Playing Capabilities of LLMs in Text-Based Virtual Worlds","date":"2024-12-07","arxiv_id":"2412.05631","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["paitesanshi/characterbox"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-the-use-of-llms-for-sql-equivalence","title":"Can the Rookies Cut the Tough Cookie? Exploring the Use of LLMs for SQL Equivalence Checking","date":"2024-12-07","arxiv_id":"2412.05561","n_code_links":0,"syntology":null},{"paper":"/paper/privagent-agentic-based-red-teaming-for-llm","slug":"privagent-agentic-based-red-teaming-for-llm","title":"PrivAgent: Agentic-based Red-teaming for LLM Privacy Leakage","date":"2024-12-07","arxiv_id":"2412.05734","n_code_links":1,"syntology":null},{"paper":null,"slug":"are-frontier-large-language-models-suitable","title":"Are Frontier Large Language Models Suitable for Q&A in Science Centres?","date":"2024-12-06","arxiv_id":"2412.05200","n_code_links":0,"syntology":null},{"paper":null,"slug":"queen-a-large-language-model-for-quechua","title":"QueEn: A Large Language Model for Quechua-English Translation","date":"2024-12-06","arxiv_id":"2412.05184","n_code_links":0,"syntology":null},{"paper":null,"slug":"discriminative-fine-tuning-of-lvlms","title":"VladVA: Discriminative Fine-tuning of LVLMs","date":"2024-12-05","arxiv_id":"2412.04378","n_code_links":0,"syntology":null},{"paper":null,"slug":"compressing-kv-cache-for-long-context-llm","title":"Compressing KV Cache for Long-Context LLM Inference with Inter-Layer Attention Similarity","date":"2024-12-03","arxiv_id":"2412.02252","n_code_links":0,"syntology":null},{"paper":"/paper/dp-2stage-adapting-language-models-as","slug":"dp-2stage-adapting-language-models-as","title":"DP-2Stage: Adapting Language Models as Differentially Private Tabular Data Generators","date":"2024-12-03","arxiv_id":"2412.02467","n_code_links":1,"syntology":null},{"paper":null,"slug":"flattering-to-deceive-the-impact-of","title":"Flattering to Deceive: The Impact of Sycophantic Behavior on User Trust in Large Language Model","date":"2024-12-03","arxiv_id":"2412.02802","n_code_links":0,"syntology":null},{"paper":null,"slug":"impact-of-data-snooping-on-deep-learning","title":"Impact of Data Snooping on Deep Learning Models for Locating Vulnerabilities in Lifted Code","date":"2024-12-03","arxiv_id":"2412.02048","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-asymptotic-behavior-of-attention-in","title":"The Asymptotic Behavior of Attention in Transformers","date":"2024-12-03","arxiv_id":"2412.02682","n_code_links":0,"syntology":null}],"record_sha256":"3bca3b1b3be7f2c13f355cc95a3686c3560185a3f2b0f284a31a0ef8054d9799","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}