{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-2/papers/4","list_of":"/method/gpt-2","method":"GPT-2","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":4,"pages_in_order":8,"rows_per_page":100,"rows":[301,400],"of":768,"counts":{"archive_papers_tagged":768,"with_a_code_link":339,"where_syntology_ran_a_sample":125,"not_listed_spam_title":0,"listed":768,"listed_where_code_ran":125,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":99,"every_run_a_failure_of_syntologys_instrument":26,"listed_with_a_run_with_no_instrument_failure":99,"listed_every_run_a_failure_of_syntologys_instrument":26,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-2","prev":"/method/gpt-2/papers/3","next":"/method/gpt-2/papers/5","papers":[{"paper":"/paper/copy-suppression-comprehensively","slug":"copy-suppression-comprehensively","title":"Copy Suppression: Comprehensively Understanding an Attention Head","date":"2023-10-06","arxiv_id":"2310.04625","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["callummcdougall/seri-mats-2023-streamlit-pages"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/nola-networks-as-linear-combination-of-low","slug":"nola-networks-as-linear-combination-of-low","title":"NOLA: Compressing LoRA using Linear Combination of Random Basis","date":"2023-10-04","arxiv_id":"2310.02556","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":2,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["UCDvision/NOLA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"polysketchformer-fast-transformers-via","title":"PolySketchFormer: Fast Transformers via Sketching Polynomial Kernels","date":"2023-10-02","arxiv_id":"2310.01655","n_code_links":0,"syntology":null},{"paper":null,"slug":"ae-gpt-using-large-language-models-to-extract","title":"AE-GPT: Using Large Language Models to Extract Adverse Events from Surveillance Reports-A Use Case with Influenza Vaccine Adverse Events","date":"2023-09-28","arxiv_id":"2309.16150","n_code_links":0,"syntology":null},{"paper":null,"slug":"constraints-first-a-new-mdd-based-model-to","title":"Constraints First: A New MDD-based Model to Generate Sentences Under Constraints","date":"2023-09-21","arxiv_id":"2309.12415","n_code_links":0,"syntology":null},{"paper":"/paper/the-languini-kitchen-enabling-language","slug":"the-languini-kitchen-enabling-language","title":"The Languini Kitchen: Enabling Language Modelling Research at Different Scales of Compute","date":"2023-09-20","arxiv_id":"2309.11197","n_code_links":1,"syntology":null},{"paper":null,"slug":"rigorously-assessing-natural-language","title":"Rigorously Assessing Natural Language Explanations of Neurons","date":"2023-09-19","arxiv_id":"2309.10312","n_code_links":0,"syntology":null},{"paper":"/paper/recap-retrieval-augmented-audio-captioning","slug":"recap-retrieval-augmented-audio-captioning","title":"RECAP: Retrieval-Augmented Audio Captioning","date":"2023-09-18","arxiv_id":"2309.09836","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["sreyan88/recap"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-modern-turkish-poet-fine-tuned-gpt-2","slug":"a-modern-turkish-poet-fine-tuned-gpt-2","title":"A Modern Turkish Poet: Fine-Tuned GPT-2","date":"2023-09-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/casteist-but-not-racist-quantifying","slug":"casteist-but-not-racist-quantifying","title":"Indian-BhED: A Dataset for Measuring India-Centric Biases in Large Language Models","date":"2023-09-15","arxiv_id":"2309.08573","n_code_links":1,"syntology":null},{"paper":"/paper/traveling-words-a-geometric-interpretation-of","slug":"traveling-words-a-geometric-interpretation-of","title":"Traveling Words: A Geometric Interpretation of Transformers","date":"2023-09-13","arxiv_id":"2309.07315","n_code_links":1,"syntology":null},{"paper":"/paper/2309-05973","slug":"2309-05973","title":"Circuit Breaking: Removing Model Behaviors with Targeted Ablation","date":"2023-09-12","arxiv_id":"2309.05973","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["xanderdavies/circuit-breaking"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"2309-06112","title":"Characterizing Latent Perspectives of Media Houses Towards Public Figures","date":"2023-09-12","arxiv_id":"2309.06112","n_code_links":0,"syntology":null},{"paper":"/paper/memory-injections-correcting-multi-hop","slug":"memory-injections-correcting-multi-hop","title":"Memory Injections: Correcting Multi-Hop Reasoning Failures during Inference in Transformer-Based Language Models","date":"2023-09-11","arxiv_id":"2309.05605","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["msakarvadia/memory_injections"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/zero-shot-audio-captioning-via-audibility","slug":"zero-shot-audio-captioning-via-audibility","title":"Zero-Shot Audio Captioning via Audibility Guidance","date":"2023-09-07","arxiv_id":"2309.03884","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-do-universal-adversarial-attacks-work-on","title":"Why do universal adversarial attacks work on large language models?: Geometry might be the answer","date":"2023-09-01","arxiv_id":"2309.00254","n_code_links":0,"syntology":null},{"paper":null,"slug":"target-independent-xla-optimization-using","title":"Target-independent XLA optimization using Reinforcement Learning","date":"2023-08-28","arxiv_id":"2308.14364","n_code_links":0,"syntology":null},{"paper":"/paper/activation-addition-steering-language-models","slug":"activation-addition-steering-language-models","title":"Steering Language Models With Activation Engineering","date":"2023-08-20","arxiv_id":"2308.10248","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["montemac/activation_additions"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-good-are-large-language-models-at-out-of","slug":"how-good-are-large-language-models-at-out-of","title":"How Good Are LLMs at Out-of-Distribution Detection?","date":"2023-08-20","arxiv_id":"2308.10261","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["awenbocc/llm-ood"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-tailored-handwritten-text-recognition","title":"A tailored Handwritten-Text-Recognition System for Medieval Latin","date":"2023-08-18","arxiv_id":"2308.09368","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-preliminary-study-on-a-conceptual-game","title":"A Preliminary Study on a Conceptual Game Feature Generation and Recommendation System","date":"2023-08-16","arxiv_id":"2308.13538","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-individual-trajectories-using-gpt","title":"Generating Individual Trajectories Using GPT-2 Trained from Scratch on Encoded Spatiotemporal Data","date":"2023-08-14","arxiv_id":"2308.07940","n_code_links":0,"syntology":null},{"paper":"/paper/audioldm-2-learning-holistic-audio-generation","slug":"audioldm-2-learning-holistic-audio-generation","title":"AudioLDM 2: Learning Holistic Audio Generation with Self-supervised Pretraining","date":"2023-08-10","arxiv_id":"2308.05734","n_code_links":2,"syntology":{"ran":16,"of":27,"n_ran_checked":16,"n_instrument":0,"unverified":11,"pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 2 honoured, 3 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","official":{"repos":["haoheliu/AudioLDM2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"i-was-a-data-augmentation-method-with-gpt-2","title":"I-WAS: a Data Augmentation Method with GPT-2 for Simile Detection","date":"2023-08-08","arxiv_id":"2308.04109","n_code_links":0,"syntology":null},{"paper":"/paper/baby-llama-knowledge-distillation-from-an","slug":"baby-llama-knowledge-distillation-from-an","title":"Baby Llama: knowledge distillation from an ensemble of teachers trained on a small dataset with no performance penalty","date":"2023-08-03","arxiv_id":"2308.02019","n_code_links":1,"syntology":null},{"paper":null,"slug":"unveiling-gender-bias-in-terms-of-profession","title":"Unveiling Gender Bias in Terms of Profession Across LLMs: Analyzing and Addressing Sociological Implications","date":"2023-07-18","arxiv_id":"2307.09162","n_code_links":0,"syntology":null},{"paper":"/paper/a-mixed-policy-to-improve-performance-of","slug":"a-mixed-policy-to-improve-performance-of","title":"A mixed policy to improve performance of language models on math problems","date":"2023-07-17","arxiv_id":"2307.08767","n_code_links":1,"syntology":null},{"paper":null,"slug":"morphpiece-moving-away-from-statistical","title":"MorphPiece : A Linguistic Tokenizer for Large Language Models","date":"2023-07-14","arxiv_id":"2307.07262","n_code_links":0,"syntology":null},{"paper":null,"slug":"simplemtod-a-simple-language-model-for","title":"SimpleMTOD: A Simple Language Model for Multimodal Task-Oriented Dialogue with Symbolic Scene Representation","date":"2023-07-10","arxiv_id":"2307.04907","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-the-efficacy-of-large-language","title":"Assessing the efficacy of large language models in generating accurate teacher responses","date":"2023-07-09","arxiv_id":"2307.04274","n_code_links":0,"syntology":null},{"paper":"/paper/came-confidence-guided-adaptive-memory","slug":"came-confidence-guided-adaptive-memory","title":"CAME: Confidence-guided Adaptive Memory Efficient Optimization","date":"2023-07-05","arxiv_id":"2307.02047","n_code_links":2,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["yangluo7/came","huawei-noah/Pretrained-Language-Model"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-the-effectiveness-of-large","title":"Evaluating the Effectiveness of Large Language Models in Representing Textual Descriptions of Geometry and Spatial Relations","date":"2023-07-05","arxiv_id":"2307.03678","n_code_links":0,"syntology":null},{"paper":null,"slug":"tensorgpt-efficient-compression-of-the","title":"TensorGPT: Efficient Compression of Large Language Models based on Tensor-Train Decomposition","date":"2023-07-02","arxiv_id":"2307.00526","n_code_links":0,"syntology":null},{"paper":"/paper/stay-on-topic-with-classifier-free-guidance","slug":"stay-on-topic-with-classifier-free-guidance","title":"Stay on topic with Classifier-Free Guidance","date":"2023-06-30","arxiv_id":"2306.17806","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-negation-detection-assessment-of-gpts","title":"A negation detection assessment of GPTs: analysis with the xNot360 dataset","date":"2023-06-29","arxiv_id":"2306.16638","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-pre-training-truly-better-than-meta","title":"Is Pre-training Truly Better Than Meta-Learning?","date":"2023-06-24","arxiv_id":"2306.13841","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-pre-trained-language-models-on","title":"Investigating Pre-trained Language Models on Cross-Domain Datasets, a Step Closer to General AI","date":"2023-06-21","arxiv_id":"2306.12205","n_code_links":0,"syntology":null},{"paper":"/paper/inrank-incremental-low-rank-learning","slug":"inrank-incremental-low-rank-learning","title":"InRank: Incremental Low-Rank Learning","date":"2023-06-20","arxiv_id":"2306.11250","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-generate-better-than-your-llm","slug":"learning-to-generate-better-than-your-llm","title":"Learning to Generate Better Than Your LLM","date":"2023-06-20","arxiv_id":"2306.11816","n_code_links":1,"syntology":null},{"paper":null,"slug":"generative-sequential-recommendation-with","title":"Generative Sequential Recommendation with GPTRec","date":"2023-06-19","arxiv_id":"2306.11114","n_code_links":0,"syntology":null},{"paper":"/paper/explore-establish-exploit-red-teaming","slug":"explore-establish-exploit-red-teaming","title":"Explore, Establish, Exploit: Red Teaming Language Models from Scratch","date":"2023-06-15","arxiv_id":"2306.09442","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["algorithmic-alignment-lab/commonclaim","thestephencasper/common_claim","thestephencasper/explore_establish_exploit_llms"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"the-pop-song-generator-designing-an-online","title":"The pop song generator: designing an online course to teach collaborative, creative AI","date":"2023-06-15","arxiv_id":"2306.10069","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-n-gram-approximation-of-pre-trained","title":"On the N-gram Approximation of Pre-trained Language Models","date":"2023-06-12","arxiv_id":"2306.06892","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-bea-2023-shared-task-on-generating-ai","title":"The BEA 2023 Shared Task on Generating AI Teacher Responses in Educational Dialogues","date":"2023-06-12","arxiv_id":"2306.06941","n_code_links":0,"syntology":null},{"paper":"/paper/language-models-can-learn-exceptions-to","slug":"language-models-can-learn-exceptions-to","title":"Language Models Can Learn Exceptions to Syntactic Rules","date":"2023-06-09","arxiv_id":"2306.05969","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-telecom-language-through-large","title":"Understanding Telecom Language Through Large Language Models","date":"2023-06-09","arxiv_id":"2306.07933","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-analysis-of-parameter-efficient","slug":"an-empirical-analysis-of-parameter-efficient","title":"An Empirical Analysis of Parameter-Efficient Methods for Debiasing Pre-Trained Language Models","date":"2023-06-06","arxiv_id":"2306.04067","n_code_links":1,"syntology":null},{"paper":null,"slug":"language-acquisition-do-children-and-language","title":"Language acquisition: do children and language models follow similar learning stages?","date":"2023-06-06","arxiv_id":"2306.03586","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-gpt-model-pre-training-using-tensor","title":"Efficient GPT Model Pre-training using Tensor Train Matrix Representation","date":"2023-06-05","arxiv_id":"2306.02697","n_code_links":0,"syntology":null},{"paper":"/paper/can-contextual-biasing-remain-effective-with","slug":"can-contextual-biasing-remain-effective-with","title":"Can Contextual Biasing Remain Effective with Whisper and GPT-2?","date":"2023-06-02","arxiv_id":"2306.01942","n_code_links":1,"syntology":null},{"paper":null,"slug":"topex-topic-based-explanations-for-model","title":"TopEx: Topic-based Explanations for Model Comparison","date":"2023-06-01","arxiv_id":"2306.00976","n_code_links":0,"syntology":null},{"paper":"/paper/test-time-training-on-nearest-neighbors-for","slug":"test-time-training-on-nearest-neighbors-for","title":"Test-Time Training on Nearest Neighbors for Large Language Models","date":"2023-05-29","arxiv_id":"2305.18466","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["socialfoundations/tttlm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"transformer-language-models-handle-word","title":"Transformer Language Models Handle Word Frequency in Prediction Head","date":"2023-05-29","arxiv_id":"2305.18294","n_code_links":0,"syntology":null},{"paper":"/paper/model-dementia-generated-data-makes-models","slug":"model-dementia-generated-data-makes-models","title":"The Curse of Recursion: Training on Generated Data Makes Models Forget","date":"2023-05-27","arxiv_id":"2305.17493","n_code_links":1,"syntology":null},{"paper":"/paper/backpack-language-models","slug":"backpack-language-models","title":"Backpack Language Models","date":"2023-05-26","arxiv_id":"2305.16765","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","official":null}},{"paper":null,"slug":"learning-and-leveraging-verifiers-to-improve","title":"Learning and Leveraging Verifiers to Improve Planning Capabilities of Pre-trained Language Models","date":"2023-05-26","arxiv_id":"2305.17077","n_code_links":0,"syntology":null},{"paper":null,"slug":"not-wacky-vs-definitely-wacky-a-study-of","title":"Not wacky vs. definitely wacky: A study of scalar adverbs in pretrained language models","date":"2023-05-25","arxiv_id":"2305.16426","n_code_links":0,"syntology":null},{"paper":"/paper/editing-commonsense-knowledge-in-gpt","slug":"editing-commonsense-knowledge-in-gpt","title":"Editing Common Sense in Transformers","date":"2023-05-24","arxiv_id":"2305.14956","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["anshitag/memit_csk"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/inference-time-policy-adapters-ipa-tailoring","slug":"inference-time-policy-adapters-ipa-tailoring","title":"Inference-Time Policy Adapters (IPA): Tailoring Extreme-Scale LMs without Fine-tuning","date":"2023-05-24","arxiv_id":"2305.15065","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":5,"n_instrument":4,"unverified":2,"pointer_only":0,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["gximinglu/ipa"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/llmdet-a-large-language-models-detection-tool","slug":"llmdet-a-large-language-models-detection-tool","title":"LLMDet: A Third Party Large Language Models Generated Text Detection Tool","date":"2023-05-24","arxiv_id":"2305.15004","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["trustedllm/llmdet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enhancing-generation-through-summarization","title":"Advancing Precise Outline-Conditioned Text Generation with Task Duality and Explicit Outline Control","date":"2023-05-23","arxiv_id":"2305.14459","n_code_links":0,"syntology":null},{"paper":"/paper/on-robustness-of-finetuned-transformer-based","slug":"on-robustness-of-finetuned-transformer-based","title":"On Robustness of Finetuned Transformer-based NLP Models","date":"2023-05-23","arxiv_id":"2305.14453","n_code_links":1,"syntology":null},{"paper":null,"slug":"probing-brain-context-sensitivity-with-masked","title":"Probing Brain Context-Sensitivity with Masked-Attention Generation","date":"2023-05-23","arxiv_id":"2305.13863","n_code_links":0,"syntology":null},{"paper":"/paper/sophia-a-scalable-stochastic-second-order","slug":"sophia-a-scalable-stochastic-second-order","title":"Sophia: A Scalable Stochastic Second-order Optimizer for Language Model Pre-training","date":"2023-05-23","arxiv_id":"2305.14342","n_code_links":7,"syntology":{"ran":13,"of":19,"n_ran_checked":12,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"13 ran (of which 4 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":null}},{"paper":"/paper/revisiting-the-architectures-like-pointer","slug":"revisiting-the-architectures-like-pointer","title":"Revisiting the Architectures like Pointer Networks to Efficiently Improve the Next Word Distribution, Summarization Factuality, and Beyond","date":"2023-05-20","arxiv_id":"2305.12289","n_code_links":1,"syntology":null},{"paper":"/paper/scaling-laws-for-language-encoding-models-in","slug":"scaling-laws-for-language-encoding-models-in","title":"Scaling laws for language encoding models in fMRI","date":"2023-05-19","arxiv_id":"2305.11863","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-reason-over-scene-graphs-a-case","title":"Learning to Reason over Scene Graphs: A Case Study of Finetuning GPT-2 into a Robot Language Model for Grounded Task Planning","date":"2023-05-12","arxiv_id":"2305.07716","n_code_links":0,"syntology":null},{"paper":"/paper/tinystories-how-small-can-language-models-be","slug":"tinystories-how-small-can-language-models-be","title":"TinyStories: How Small Can Language Models Be and Still Speak Coherent English?","date":"2023-05-12","arxiv_id":"2305.07759","n_code_links":8,"syntology":{"ran":10,"of":18,"n_ran_checked":8,"n_instrument":2,"unverified":8,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","official":null}},{"paper":null,"slug":"investigating-the-effect-of-sub-word","title":"Effects of sub-word segmentation on performance of transformer language models","date":"2023-05-09","arxiv_id":"2305.05480","n_code_links":0,"syntology":null},{"paper":"/paper/neurocomparatives-neuro-symbolic-distillation","slug":"neurocomparatives-neuro-symbolic-distillation","title":"NeuroComparatives: Neuro-Symbolic Distillation of Comparative Knowledge","date":"2023-05-08","arxiv_id":"2305.04978","n_code_links":1,"syntology":null},{"paper":null,"slug":"adapting-transformer-language-models-for","title":"Adapting Transformer Language Models for Predictive Typing in Brain-Computer Interfaces","date":"2023-05-05","arxiv_id":"2305.03819","n_code_links":0,"syntology":null},{"paper":"/paper/how-does-gpt-2-compute-greater-than-1","slug":"how-does-gpt-2-compute-greater-than-1","title":"How does GPT-2 compute greater-than?: Interpreting mathematical abilities in a pre-trained language model","date":"2023-04-30","arxiv_id":"2305.00586","n_code_links":3,"syntology":null},{"paper":"/paper/towards-automated-circuit-discovery-for-1","slug":"towards-automated-circuit-discovery-for-1","title":"Towards Automated Circuit Discovery for Mechanistic Interpretability","date":"2023-04-28","arxiv_id":"2304.14997","n_code_links":4,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["arthurconmy/automatic-circuit-discovery","neelnanda-io/transformerlens"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"how-to-do-things-with-deep-learning-code","title":"How to Do Things with Deep Learning Code","date":"2023-04-19","arxiv_id":"2304.09406","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-multitask-learning-to","slug":"an-empirical-study-of-multitask-learning-to","title":"An Empirical Study of Multitask Learning to Improve Open Domain Dialogue Systems","date":"2023-04-17","arxiv_id":"2304.08115","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-chatgpt-forecast-stock-price-movements","title":"Can ChatGPT Forecast Stock Price Movements? Return Predictability and Large Language Models","date":"2023-04-15","arxiv_id":"2304.07619","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-code-generation","title":"Stochastic Code Generation","date":"2023-04-14","arxiv_id":"2304.08243","n_code_links":0,"syntology":null},{"paper":"/paper/pgtask-introducing-the-task-of-profile","slug":"pgtask-introducing-the-task-of-profile","title":"PGTask: Introducing the Task of Profile Generation from Dialogues","date":"2023-04-13","arxiv_id":"2304.06634","n_code_links":1,"syntology":null},{"paper":"/paper/localizing-model-behavior-with-path-patching","slug":"localizing-model-behavior-with-path-patching","title":"Localizing Model Behavior with Path Patching","date":"2023-04-12","arxiv_id":"2304.05969","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["redwoodresearch/rust_circuit_public"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-the-possibilities-of-ai-generated-text","title":"On the Possibilities of AI-Generated Text Detection","date":"2023-04-10","arxiv_id":"2304.04736","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt4rec-a-generative-framework-for","title":"GPT4Rec: A Generative Framework for Personalized Recommendation and User Interests Interpretation","date":"2023-04-08","arxiv_id":"2304.03879","n_code_links":0,"syntology":null},{"paper":"/paper/llmmaps-a-visual-metaphor-for-stratified","slug":"llmmaps-a-visual-metaphor-for-stratified","title":"LLMMaps -- A Visual Metaphor for Stratified Evaluation of Large Language Models","date":"2023-04-02","arxiv_id":"2304.00457","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-do-decoding-algorithms-distribute","title":"How do decoding algorithms distribute information in dialogue responses?","date":"2023-03-29","arxiv_id":"2303.17006","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-generalizable-end-to-end-task","slug":"zero-shot-generalizable-end-to-end-task","title":"Zero-Shot Generalizable End-to-End Task-Oriented Dialog System using Context Summarization and Domain Schema","date":"2023-03-28","arxiv_id":"2303.16252","n_code_links":1,"syntology":null},{"paper":null,"slug":"personalizing-task-oriented-dialog-systems","title":"Personalizing Task-oriented Dialog Systems via Zero-shot Generalizable Reward Function","date":"2023-03-24","arxiv_id":"2303.13797","n_code_links":0,"syntology":null},{"paper":"/paper/jump-to-conclusions-short-cutting","slug":"jump-to-conclusions-short-cutting","title":"Jump to Conclusions: Short-Cutting Transformers With Linear Transformations","date":"2023-03-16","arxiv_id":"2303.09435","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sashayd/mat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gcre-gpt-a-generative-model-for-comparative","title":"GCRE-GPT: A Generative Model for Comparative Relation Extraction","date":"2023-03-15","arxiv_id":"2303.08601","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-combinatorial-prompts-for-universal","title":"Learning Combinatorial Prompts for Universal Controllable Image Captioning","date":"2023-03-11","arxiv_id":"2303.06338","n_code_links":0,"syntology":null},{"paper":"/paper/open-ended-medical-visual-question-answering","slug":"open-ended-medical-visual-question-answering","title":"Open-Ended Medical Visual Question Answering Through Prefix Tuning of Language Models","date":"2023-03-10","arxiv_id":"2303.05977","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tjvsonsbeek/open-ended-medical-vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-the-risks-of-stealing-the-decoding","slug":"on-the-risks-of-stealing-the-decoding","title":"Stealing the Decoding Algorithms of Language Models","date":"2023-03-08","arxiv_id":"2303.04729","n_code_links":1,"syntology":null},{"paper":"/paper/towards-zero-shot-functional-compositionality","slug":"towards-zero-shot-functional-compositionality","title":"Towards Zero-Shot Functional Compositionality of Language Models","date":"2023-03-06","arxiv_id":"2303.03103","n_code_links":1,"syntology":null},{"paper":null,"slug":"fqp-2-0-industry-trend-analysis-via","title":"Industry Risk Assessment via Hierarchical Financial Data Using Stock Market Sentiment Indicators","date":"2023-03-05","arxiv_id":"2303.02707","n_code_links":0,"syntology":null},{"paper":"/paper/information-restricted-neural-language-models","slug":"information-restricted-neural-language-models","title":"Information-Restricted Neural Language Models Reveal Different Brain Regions' Sensitivity to Semantics, Syntax and Context","date":"2023-02-28","arxiv_id":"2302.14389","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alexandrepsq/information-restrited-nlms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/inseq-an-interpretability-toolkit-for","slug":"inseq-an-interpretability-toolkit-for","title":"Inseq: An Interpretability Toolkit for Sequence Generation Models","date":"2023-02-27","arxiv_id":"2302.13942","n_code_links":2,"syntology":null},{"paper":null,"slug":"fast-attention-requires-bounded-entries","title":"Fast Attention Requires Bounded Entries","date":"2023-02-26","arxiv_id":"2302.13214","n_code_links":0,"syntology":null},{"paper":"/paper/conveying-the-predicted-future-to-users-a","slug":"conveying-the-predicted-future-to-users-a","title":"Conveying the Predicted Future to Users: A Case Study of Story Plot Prediction","date":"2023-02-17","arxiv_id":"2302.09122","n_code_links":1,"syntology":null},{"paper":"/paper/tree-based-representation-and-generation-of","slug":"tree-based-representation-and-generation-of","title":"Tree-Based Representation and Generation of Natural and Mathematical Language","date":"2023-02-15","arxiv_id":"2302.07974","n_code_links":1,"syntology":null},{"paper":"/paper/fairpy-a-toolkit-for-evaluation-of-social","slug":"fairpy-a-toolkit-for-evaluation-of-social","title":"FairPy: A Toolkit for Evaluation of Prediction Biases and their Mitigation in Large Language Models","date":"2023-02-10","arxiv_id":"2302.05508","n_code_links":1,"syntology":null},{"paper":null,"slug":"nationality-bias-in-text-generation","title":"Nationality Bias in Text Generation","date":"2023-02-05","arxiv_id":"2302.02463","n_code_links":0,"syntology":null},{"paper":"/paper/realtabformer-generating-realistic-relational","slug":"realtabformer-generating-realistic-relational","title":"REaLTabFormer: Generating Realistic Relational and Tabular Data using Transformers","date":"2023-02-04","arxiv_id":"2302.02041","n_code_links":3,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["avsolatorio/realtabformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"c5920f0750a49dc88b5a16ef25d0f7395a8c0e04bfefddd27c5242dafa0579f8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}