{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-2/papers/2","list_of":"/method/gpt-2","method":"GPT-2","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":8,"rows_per_page":100,"rows":[101,200],"of":768,"counts":{"archive_papers_tagged":768,"with_a_code_link":339,"where_syntology_ran_a_sample":125,"not_listed_spam_title":0,"listed":768,"listed_where_code_ran":125,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":99,"every_run_a_failure_of_syntologys_instrument":26,"listed_with_a_run_with_no_instrument_failure":99,"listed_every_run_a_failure_of_syntologys_instrument":26,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-2","prev":"/method/gpt-2","next":"/method/gpt-2/papers/3","papers":[{"paper":null,"slug":"improving-next-tokens-via-second-last","title":"Improving Next Tokens via Second-Last Predictions with Generate and Refine","date":"2024-11-23","arxiv_id":"2411.15661","n_code_links":0,"syntology":null},{"paper":"/paper/mars-unleashing-the-power-of-variance","slug":"mars-unleashing-the-power-of-variance","title":"MARS: Unleashing the Power of Variance Reduction for Training Large Models","date":"2024-11-15","arxiv_id":"2411.10438","n_code_links":2,"syntology":{"ran":3,"of":5,"n_ran_checked":2,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["AGI-Arena/MARS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"take-package-as-language-anomaly-detection","title":"Take Package as Language: Anomaly Detection Using Transformer","date":"2024-11-15","arxiv_id":"2412.04473","n_code_links":0,"syntology":null},{"paper":null,"slug":"babylm-challenge-exploring-the-effect-of","title":"BabyLM Challenge: Exploring the Effect of Variation Sets on Language Model Training Efficiency","date":"2024-11-14","arxiv_id":"2411.09587","n_code_links":0,"syntology":null},{"paper":"/paper/autonomous-droplet-microfluidic-design","slug":"autonomous-droplet-microfluidic-design","title":"Autonomous Droplet Microfluidic Design Framework with Large Language Models","date":"2024-11-11","arxiv_id":"2411.06691","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-active-privacy-auditing-in-supervised-fine","title":"On Active Privacy Auditing in Supervised Fine-tuning for White-Box Language Models","date":"2024-11-11","arxiv_id":"2411.07070","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-efficient-fine-tuning-for-gpt-like","title":"Prompt-Efficient Fine-Tuning for GPT-like Deep Models to Reduce Hallucination and to Improve Reproducibility in Scientific Text Generation Using Stochastic Optimisation Techniques","date":"2024-11-10","arxiv_id":"2411.06445","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-robustness-of-in-context-learning","title":"Adversarial Robustness of In-Context Learning in Transformers for Linear Regression","date":"2024-11-07","arxiv_id":"2411.05189","n_code_links":0,"syntology":null},{"paper":"/paper/can-custom-models-learn-in-context-an","slug":"can-custom-models-learn-in-context-an","title":"Can Custom Models Learn In-Context? An Exploration of Hybrid Architecture Performance on In-Context Learning Tasks","date":"2024-11-06","arxiv_id":"2411.03945","n_code_links":1,"syntology":null},{"paper":"/paper/towards-interpreting-language-models-a-case","slug":"towards-interpreting-language-models-a-case","title":"Towards Interpreting Language Models: A Case Study in Multi-Hop Reasoning","date":"2024-11-06","arxiv_id":"2411.05037","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["msakarvadia/attentionlens"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enriching-tabular-data-with-contextual-llm","title":"Enriching Tabular Data with Contextual LLM Embeddings: A Comprehensive Ablation Study for Ensemble Classifiers","date":"2024-11-03","arxiv_id":"2411.01645","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-neural-network-interpretability-1","slug":"enhancing-neural-network-interpretability-1","title":"Enhancing Neural Network Interpretability with Feature-Aligned Sparse Autoencoders","date":"2024-11-02","arxiv_id":"2411.01220","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["luke-marks0/mutual-feature-regularization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/blast-block-level-adaptive-structured","slug":"blast-block-level-adaptive-structured","title":"BLAST: Block-Level Adaptive Structured Matrices for Efficient Deep Neural Network Inference","date":"2024-10-28","arxiv_id":"2410.21262","n_code_links":1,"syntology":null},{"paper":null,"slug":"causal-interventions-on-causal-paths-mapping","title":"Causal Interventions on Causal Paths: Mapping GPT-2's Reasoning From Syntax to Semantics","date":"2024-10-28","arxiv_id":"2410.21353","n_code_links":0,"syntology":null},{"paper":"/paper/multitok-variable-length-tokenization-for","slug":"multitok-variable-length-tokenization-for","title":"MultiTok: Variable-Length Tokenization for Efficient LLMs Adapted from LZW Compression","date":"2024-10-28","arxiv_id":"2410.21548","n_code_links":1,"syntology":null},{"paper":"/paper/scaling-up-masked-diffusion-models-on-text","slug":"scaling-up-masked-diffusion-models-on-text","title":"Scaling up Masked Diffusion Models on Text","date":"2024-10-24","arxiv_id":"2410.18514","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ml-gsai/smdm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/differentially-private-learning-needs-better","slug":"differentially-private-learning-needs-better","title":"Differentially Private Learning Needs Better Model Initialization and Self-Distillation","date":"2024-10-23","arxiv_id":"2410.17566","n_code_links":1,"syntology":null},{"paper":"/paper/dnahlm-dna-sequence-and-human-language-mixed","slug":"dnahlm-dna-sequence-and-human-language-mixed","title":"DNAHLM -- DNA sequence and Human Language mixed large language Model","date":"2024-10-22","arxiv_id":"2410.16917","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-possibilities-of-ai-powered-legal","slug":"exploring-possibilities-of-ai-powered-legal","title":"Exploring Possibilities of AI-Powered Legal Assistance in Bangladesh through Large Language Modeling","date":"2024-10-22","arxiv_id":"2410.17210","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-neuron-level-interpretability-with","title":"Improving Neuron-level Interpretability with White-box Language Models","date":"2024-10-21","arxiv_id":"2410.16443","n_code_links":0,"syntology":null},{"paper":null,"slug":"bias-amplification-language-models-as","title":"Bias Amplification: Language Models as Increasingly Biased Media","date":"2024-10-19","arxiv_id":"2410.15234","n_code_links":0,"syntology":null},{"paper":null,"slug":"training-compute-optimal-vision-transformers","title":"Training Compute-Optimal Vision Transformers for Brain Encoding","date":"2024-10-17","arxiv_id":"2410.19810","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-scaling-versus-task-scaling-in-in","title":"Context-Scaling versus Task-Scaling in In-Context Learning","date":"2024-10-16","arxiv_id":"2410.12783","n_code_links":0,"syntology":null},{"paper":null,"slug":"fusionllm-a-decentralized-llm-training-system","title":"FusionLLM: A Decentralized LLM Training System on Geo-distributed GPUs with Adaptive Compression","date":"2024-10-16","arxiv_id":"2410.12707","n_code_links":0,"syntology":null},{"paper":null,"slug":"kallini-et-al-2024-do-not-compare-impossible","title":"Kallini et al. (2024) do not compare impossible languages with constituency-based ones","date":"2024-10-16","arxiv_id":"2410.12271","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-in-context-learning-really-generalize-to","title":"Can In-context Learning Really Generalize to Out-of-distribution Tasks?","date":"2024-10-13","arxiv_id":"2410.09695","n_code_links":0,"syntology":null},{"paper":"/paper/adam-exploits-ell-infty-geometry-of-loss","slug":"adam-exploits-ell-infty-geometry-of-loss","title":"Adam Exploits $\\ell_\\infty$-geometry of Loss Landscape via Coordinate-wise Adaptivity","date":"2024-10-10","arxiv_id":"2410.08198","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":6,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["mohamad-amin/adam-coordinate-adaptivity"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sage-scalable-ground-truth-evaluations-for","title":"SAGE: Scalable Ground Truth Evaluations for Large Sparse Autoencoders","date":"2024-10-09","arxiv_id":"2410.07456","n_code_links":0,"syntology":null},{"paper":"/paper/a-second-order-like-optimizer-with-adaptive","slug":"a-second-order-like-optimizer-with-adaptive","title":"A second-order-like optimizer with adaptive gradient scaling for deep learning","date":"2024-10-08","arxiv_id":"2410.05871","n_code_links":1,"syntology":null},{"paper":"/paper/coevolving-with-the-other-you-fine-tuning-llm","slug":"coevolving-with-the-other-you-fine-tuning-llm","title":"Coevolving with the Other You: Fine-Tuning LLM with Sequential Cooperative Multi-Agent Reinforcement Learning","date":"2024-10-08","arxiv_id":"2410.06101","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":6,"n_instrument":2,"unverified":3,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["Harry67Hu/CORY"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lpzero-language-model-zero-cost-proxy-search","title":"LPZero: Language Model Zero-cost Proxy Search from Zero","date":"2024-10-07","arxiv_id":"2410.04808","n_code_links":0,"syntology":null},{"paper":"/paper/how-language-models-prioritize-contextual","slug":"how-language-models-prioritize-contextual","title":"How Language Models Prioritize Contextual Grammatical Cues?","date":"2024-10-04","arxiv_id":"2410.03447","n_code_links":1,"syntology":null},{"paper":"/paper/sparse-attention-decomposition-applied-to","slug":"sparse-attention-decomposition-applied-to","title":"Sparse Attention Decomposition Applied to Circuit Tracing","date":"2024-10-01","arxiv_id":"2410.00340","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-the-fairness-of-task-adaptive","slug":"evaluating-the-fairness-of-task-adaptive","title":"Evaluating the fairness of task-adaptive pretraining on unlabeled test data before few-shot text classification","date":"2024-09-30","arxiv_id":"2410.00179","n_code_links":1,"syntology":null},{"paper":null,"slug":"modelando-procesos-cognitivos-de-la-lectura","title":"Modelando procesos cognitivos de la lectura natural con GPT-2","date":"2024-09-30","arxiv_id":"2409.20174","n_code_links":0,"syntology":null},{"paper":"/paper/analog-in-memory-computing-attention","slug":"analog-in-memory-computing-attention","title":"Analog In-Memory Computing Attention Mechanism for Fast and Energy-Efficient Large Language Models","date":"2024-09-28","arxiv_id":"2409.19315","n_code_links":1,"syntology":null},{"paper":null,"slug":"experimental-evaluation-of-machine-learning","title":"Experimental Evaluation of Machine Learning Models for Goal-oriented Customer Service Chatbot with Pipeline Architecture","date":"2024-09-27","arxiv_id":"2409.18568","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-unidirectional-bidirectional-and","title":"Comparing Unidirectional, Bidirectional, and Word2vec Models for Discovering Vulnerabilities in Compiled Lifted Code","date":"2024-09-26","arxiv_id":"2409.17513","n_code_links":0,"syntology":null},{"paper":"/paper/sdba-a-stealthy-and-long-lasting-durable","slug":"sdba-a-stealthy-and-long-lasting-durable","title":"SDBA: A Stealthy and Long-Lasting Durable Backdoor Attack in Federated Learning","date":"2024-09-23","arxiv_id":"2409.14805","n_code_links":1,"syntology":null},{"paper":null,"slug":"drift-to-remember","title":"Drift to Remember","date":"2024-09-21","arxiv_id":"2409.13997","n_code_links":0,"syntology":null},{"paper":"/paper/loop-residual-neural-networks-for-iterative","slug":"loop-residual-neural-networks-for-iterative","title":"Loop Neural Networks for Parameter Sharing","date":"2024-09-21","arxiv_id":"2409.14199","n_code_links":0,"syntology":null},{"paper":null,"slug":"hut-a-more-computation-efficient-fine-tuning","title":"HUT: A More Computation Efficient Fine-Tuning Method With Hadamard Updated Transformation","date":"2024-09-20","arxiv_id":"2409.13501","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-unified-framework-to-classify-business","title":"A Unified Framework to Classify Business Activities into International Standard Industrial Classification through Large Language Models for Circular Economy","date":"2024-09-17","arxiv_id":"2409.18988","n_code_links":0,"syntology":null},{"paper":null,"slug":"stable-language-model-pre-training-by","title":"Stable Language Model Pre-training by Reducing Embedding Variability","date":"2024-09-12","arxiv_id":"2409.07787","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-large-language-model-pretraining","title":"Accelerating Large Language Model Pretraining via LFR Pedagogy: Learn, Focus, and Review","date":"2024-09-10","arxiv_id":"2409.06131","n_code_links":0,"syntology":null},{"paper":null,"slug":"bypassing-darcy-defense-indistinguishable","title":"Bypassing DARCY Defense: Indistinguishable Universal Adversarial Triggers","date":"2024-09-05","arxiv_id":"2409.03183","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-open-source-sparse-autoencoders-on","slug":"evaluating-open-source-sparse-autoencoders-on","title":"Evaluating Open-Source Sparse Autoencoders on Disentangling Factual Knowledge in GPT-2 Small","date":"2024-09-05","arxiv_id":"2409.04478","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["maheepchaudhary/sae-ravel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"retrieval-augmented-natural-language","title":"Retrieval-Augmented Natural Language Reasoning for Explainable Visual Question Answering","date":"2024-08-30","arxiv_id":"2408.17006","n_code_links":0,"syntology":null},{"paper":"/paper/llava-chef-a-multi-modal-generative-model-for","slug":"llava-chef-a-multi-modal-generative-model-for","title":"LLaVA-Chef: A Multi-modal Generative Model for Food Recipes","date":"2024-08-29","arxiv_id":"2408.16889","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-multi-hop-reasoning-through","title":"Enhancing Multi-hop Reasoning through Knowledge Erasure in Large Language Model Editing","date":"2024-08-22","arxiv_id":"2408.12456","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixed-sparsity-training-achieving-4-times","title":"Mixed Sparsity Training: Achieving 4$\\times$ FLOP Reduction for Transformer Pretraining","date":"2024-08-21","arxiv_id":"2408.11746","n_code_links":0,"syntology":null},{"paper":null,"slug":"tracing-privacy-leakage-of-language-models-to","title":"Tracing Privacy Leakage of Language Models to Training Data via Adjusted Influence Functions","date":"2024-08-20","arxiv_id":"2408.10468","n_code_links":0,"syntology":null},{"paper":"/paper/enhance-lifelong-model-editing-with","slug":"enhance-lifelong-model-editing-with","title":"ELDER: Enhancing Lifelong Model Editing with Mixture-of-LoRA","date":"2024-08-19","arxiv_id":"2408.11869","n_code_links":1,"syntology":null},{"paper":null,"slug":"pragmatic-inference-of-scalar-implicature-by","title":"Pragmatic inference of scalar implicature by LLMs","date":"2024-08-13","arxiv_id":"2408.06673","n_code_links":0,"syntology":null},{"paper":null,"slug":"retrieval-augmented-code-completion-for-local","title":"Retrieval-augmented code completion for local projects using large language models","date":"2024-08-09","arxiv_id":"2408.05026","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-explainer-interactive-learning-of","slug":"transformer-explainer-interactive-learning-of","title":"Transformer Explainer: Interactive Learning of Text-Generative Models","date":"2024-08-08","arxiv_id":"2408.04619","n_code_links":1,"syntology":null},{"paper":"/paper/image-to-latex-converter-for-mathematical","slug":"image-to-latex-converter-for-mathematical","title":"Image-to-LaTeX Converter for Mathematical Formulas and Text","date":"2024-08-07","arxiv_id":"2408.04015","n_code_links":1,"syntology":null},{"paper":"/paper/is-child-directed-speech-effective-training","slug":"is-child-directed-speech-effective-training","title":"Is Child-Directed Speech Effective Training Data for Language Models?","date":"2024-08-07","arxiv_id":"2408.03617","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":5,"n_instrument":2,"unverified":2,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["styfeng/tinydialogues"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"socfedgpt-federated-gpt-based-adaptive","title":"SocFedGPT: Federated GPT-based Adaptive Content Filtering System Leveraging User Interactions in Social Networks","date":"2024-08-07","arxiv_id":"2408.05243","n_code_links":0,"syntology":null},{"paper":"/paper/trafficgpt-an-llm-approach-for-open-set","slug":"trafficgpt-an-llm-approach-for-open-set","title":"TrafficGPT: An LLM Approach for Open-Set Encrypted Traffic Classification","date":"2024-08-06","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/2408-01966","slug":"2408-01966","title":"ML-EAT: A Multilevel Embedding Association Test for Interpretable and Transparent Social Science","date":"2024-08-04","arxiv_id":"2408.01966","n_code_links":1,"syntology":null},{"paper":"/paper/autoscale-automatic-prediction-of-compute","slug":"autoscale-automatic-prediction-of-compute","title":"AutoScale: Scale-Aware Data Mixing for Pre-Training LLMs","date":"2024-07-29","arxiv_id":"2407.20177","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["feiyang-k/autoscale"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/detecting-and-understanding-vulnerabilities","slug":"detecting-and-understanding-vulnerabilities","title":"Detecting and Understanding Vulnerabilities in Language Models via Mechanistic Interpretability","date":"2024-07-29","arxiv_id":"2407.19842","n_code_links":1,"syntology":null},{"paper":null,"slug":"mechanistic-interpretability-of-large","title":"Mechanistic interpretability of large language models with applications to the financial services industry","date":"2024-07-15","arxiv_id":"2407.11215","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-in-store-customer-journeys-from","title":"Generating In-store Customer Journeys from Scratch with GPT Architectures","date":"2024-07-13","arxiv_id":"2407.11081","n_code_links":0,"syntology":null},{"paper":"/paper/astprompter-weakly-supervised-automated","slug":"astprompter-weakly-supervised-automated","title":"ASTPrompter: Weakly Supervised Automated Language Model Red-Teaming to Identify Low-Perplexity Toxic Prompts","date":"2024-07-12","arxiv_id":"2407.09447","n_code_links":1,"syntology":null},{"paper":"/paper/rosa-random-subspace-adaptation-for-efficient","slug":"rosa-random-subspace-adaptation-for-efficient","title":"ROSA: Random Subspace Adaptation for Efficient Fine-Tuning","date":"2024-07-10","arxiv_id":"2407.07802","n_code_links":1,"syntology":null},{"paper":null,"slug":"raply-a-profanity-mitigated-rap-generator","title":"Raply: A profanity-mitigated rap generator","date":"2024-07-09","arxiv_id":"2407.06941","n_code_links":0,"syntology":null},{"paper":null,"slug":"slice-100k-a-multimodal-dataset-for-extrusion","title":"Slice-100K: A Multimodal Dataset for Extrusion-based 3D Printing","date":"2024-07-04","arxiv_id":"2407.04180","n_code_links":0,"syntology":null},{"paper":null,"slug":"obfuscatune-obfuscated-offsite-fine-tuning","title":"ObfuscaTune: Obfuscated Offsite Fine-tuning and Inference of Proprietary LLMs on Private Datasets","date":"2024-07-03","arxiv_id":"2407.02960","n_code_links":0,"syntology":null},{"paper":"/paper/parm-efficient-training-of-large-sparsely","slug":"parm-efficient-training-of-large-sparsely","title":"Parm: Efficient Training of Large Sparsely-Activated Models with Dedicated Schedules","date":"2024-06-30","arxiv_id":"2407.00599","n_code_links":1,"syntology":null},{"paper":"/paper/machine-learning-predictors-for-min-entropy","slug":"machine-learning-predictors-for-min-entropy","title":"Machine Learning Predictors for Min-Entropy Estimation","date":"2024-06-28","arxiv_id":"2406.19983","n_code_links":1,"syntology":null},{"paper":null,"slug":"scalebio-scalable-bilevel-optimization-for","title":"ScaleBiO: Scalable Bilevel Optimization for LLM Data Reweighting","date":"2024-06-28","arxiv_id":"2406.19976","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuned-network-relies-on-generic","title":"Fine-tuned network relies on generic representation to solve unseen cognitive task","date":"2024-06-27","arxiv_id":"2406.18926","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-entity-recognition-using-ensembles","title":"Improving Entity Recognition Using Ensembles of Deep Learning and Fine-tuned Large Language Models: A Case Study on Adverse Event Extraction from Multiple Sources","date":"2024-06-26","arxiv_id":"2406.18049","n_code_links":0,"syntology":null},{"paper":"/paper/interpreting-attention-layer-outputs-with","slug":"interpreting-attention-layer-outputs-with","title":"Interpreting Attention Layer Outputs with Sparse Autoencoders","date":"2024-06-25","arxiv_id":"2406.17759","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["ckkissane/attention-output-saes"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"what-do-the-circuits-mean-a-knowledge-edit","title":"Understanding Language Model Circuits through Knowledge Editing","date":"2024-06-25","arxiv_id":"2406.17241","n_code_links":0,"syntology":null},{"paper":"/paper/finding-transformer-circuits-with-edge","slug":"finding-transformer-circuits-with-edge","title":"Finding Transformer Circuits with Edge Pruning","date":"2024-06-24","arxiv_id":"2406.16778","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":3,"n_instrument":5,"unverified":0,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["princeton-nlp/edge-pruning"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"anime-popularity-prediction-before-huge","title":"Anime Popularity Prediction Before Huge Investments: a Multimodal Approach Using Deep Learning","date":"2024-06-21","arxiv_id":"2406.16961","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-makes-two-models-think-alike","title":"What Makes Two Language Models Think Alike?","date":"2024-06-18","arxiv_id":"2406.12620","n_code_links":0,"syntology":null},{"paper":null,"slug":"promises-outlooks-and-challenges-of-diffusion","title":"Promises, Outlooks and Challenges of Diffusion Language Modeling","date":"2024-06-17","arxiv_id":"2406.11473","n_code_links":0,"syntology":null},{"paper":"/paper/sharelora-parameter-efficient-and-robust","slug":"sharelora-parameter-efficient-and-robust","title":"ShareLoRA: Parameter Efficient and Robust Large Language Model Fine-tuning via Shared Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.10785","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":0,"n_instrument":5,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Rain9876/ShareLoRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/mint-a-multi-modal-image-and-narrative-text","slug":"mint-a-multi-modal-image-and-narrative-text","title":"MINT: a Multi-modal Image and Narrative Text Dubbing Dataset for Foley Audio Content Planning and Generation","date":"2024-06-15","arxiv_id":"2406.10591","n_code_links":1,"syntology":null},{"paper":"/paper/towards-efficient-pareto-set-approximation","slug":"towards-efficient-pareto-set-approximation","title":"Towards Efficient Pareto Set Approximation via Mixture of Experts Based Model Fusion","date":"2024-06-14","arxiv_id":"2406.09770","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-more-practical-approach-to-machine","title":"A More Practical Approach to Machine Unlearning","date":"2024-06-13","arxiv_id":"2406.09391","n_code_links":0,"syntology":null},{"paper":null,"slug":"talking-heads-understanding-inter-layer","title":"Talking Heads: Understanding Inter-layer Communication in Transformer Language Models","date":"2024-06-13","arxiv_id":"2406.09519","n_code_links":0,"syntology":null},{"paper":"/paper/compute-better-spent-replacing-dense-layers","slug":"compute-better-spent-replacing-dense-layers","title":"Compute Better Spent: Replacing Dense Layers with Structured Matrices","date":"2024-06-10","arxiv_id":"2406.06248","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shikaiqiu/compute-better-spent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"critical-phase-transition-in-a-large-language","title":"Critical Phase Transition in Large Language Models","date":"2024-06-08","arxiv_id":"2406.05335","n_code_links":0,"syntology":null},{"paper":null,"slug":"vtrans-accelerating-transformer-compression","title":"VTrans: Accelerating Transformer Compression with Variational Information Bottleneck based Pruning","date":"2024-06-07","arxiv_id":"2406.05276","n_code_links":0,"syntology":null},{"paper":"/paper/simplified-and-generalized-masked-diffusion","slug":"simplified-and-generalized-masked-diffusion","title":"Simplified and Generalized Masked Diffusion for Discrete Data","date":"2024-06-06","arxiv_id":"2406.04329","n_code_links":1,"syntology":{"ran":6,"of":21,"n_ran_checked":2,"n_instrument":4,"unverified":15,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 15 unverified","official":{"repos":["google-deepmind/md4"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":15,"ran_from_kinds":["official"]}}},{"paper":"/paper/your-absorbing-discrete-diffusion-secretly","slug":"your-absorbing-discrete-diffusion-secretly","title":"Your Absorbing Discrete Diffusion Secretly Models the Conditional Distributions of Clean Data","date":"2024-06-06","arxiv_id":"2406.03736","n_code_links":2,"syntology":{"ran":13,"of":13,"n_ran_checked":12,"n_instrument":1,"unverified":0,"pointer_only":13,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ml-gsai/radd"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exact-conversion-of-in-context-learning-to","title":"Exact Conversion of In-Context Learning to Model Weights in Linearized-Attention Transformers","date":"2024-06-05","arxiv_id":"2406.02847","n_code_links":0,"syntology":null},{"paper":"/paper/rico-reddit-ideological-communities","slug":"rico-reddit-ideological-communities","title":"RICo: Reddit ideological communities","date":"2024-06-05","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/too-big-to-fail-larger-language-models-are","slug":"too-big-to-fail-larger-language-models-are","title":"Too Big to Fail: Larger Language Models are Disproportionately Resilient to Induction of Dementia-Related Linguistic Anomalies","date":"2024-06-05","arxiv_id":"2406.02830","n_code_links":1,"syntology":null},{"paper":null,"slug":"in-context-learning-of-physical-properties","title":"In-Context Learning of Physical Properties: Few-Shot Adaptation to Out-of-Distribution Molecular Graphs","date":"2024-06-03","arxiv_id":"2406.01808","n_code_links":0,"syntology":null},{"paper":null,"slug":"lolameme-logic-language-memory-mechanistic","title":"LOLAMEME: Logic, Language, Memory, Mechanistic Framework","date":"2024-05-31","arxiv_id":"2406.02592","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-graph-tuning-real-time-large","title":"Knowledge Graph Tuning: Real-time Large Language Model Personalization based on Human Feedback","date":"2024-05-30","arxiv_id":"2405.19686","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-model-agnostic-alignment-via","title":"Efficient Model-agnostic Alignment via Bayesian Persuasion","date":"2024-05-29","arxiv_id":"2405.18718","n_code_links":0,"syntology":null},{"paper":null,"slug":"lmo-dp-optimizing-the-randomization-mechanism","title":"LMO-DP: Optimizing the Randomization Mechanism for Differentially Private Fine-Tuning (Large) Language Models","date":"2024-05-29","arxiv_id":"2405.18776","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-ppo-ed-language-models-hackable","title":"Are PPO-ed Language Models Hackable?","date":"2024-05-28","arxiv_id":"2406.02577","n_code_links":0,"syntology":null}],"record_sha256":"8b4d0621548e20669366dd0f6ae52786aabcf1c7f187f817b58327c7cf1047f8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}