{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/gelu-fast","entry":"gelu_fast","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":39,"n_papers_ran":22,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":4,"n_samples_ran":1,"n_samples_fingerprinted":1,"n_places":45,"n_places_pointer_only":25,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":0,"unverified":3},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2410.19123","paper":"/paper/read-me-refactorizing-llms-as-router","title":"Read-ME: Refactorizing LLMs as Router-Decoupled Mixture of Experts with System Co-Design","date":"2024-10-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"VITA-Group/READ-ME","path":"src/transformers/activations_tf.py","file_url":"https://github.com/VITA-Group/READ-ME/blob/HEAD/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2410.12178","paper":"/paper/model-balancing-helps-low-data-training-and","title":"Model Balancing Helps Low-data Training and Fine-tuning","date":"2024-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zihanghliu/modelbalancing","path":"transformers/src/transformers/activations_tf.py","file_url":"https://github.com/zihanghliu/modelbalancing/blob/HEAD/transformers/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2410.05748","paper":"/paper/label-confidence-weighted-learning-for-target","title":"Label Confidence Weighted Learning for Target-level Sentence Simplification","date":"2024-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"astro-jon/LCWL","path":"src/transformers/activations.py","file_url":"https://github.com/astro-jon/LCWL/blob/HEAD/src/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2409.15647","paper":"/paper/looped-transformers-for-length-generalization","title":"Looped Transformers for Length Generalization","date":"2024-09-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"UW-Madison-Lee-Lab/looped-tf","path":"src/transformers/activations_tf.py","file_url":"https://github.com/UW-Madison-Lee-Lab/looped-tf/blob/HEAD/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2407.07071","paper":"/paper/lookback-lens-detecting-and-mitigating","title":"Lookback Lens: Detecting and Mitigating Contextual Hallucinations in Large Language Models Using Only Attention Maps","date":"2024-07-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"voidism/lookback-lens","path":"transformers-4.32.0/src/transformers/activations_tf.py","file_url":"https://github.com/voidism/lookback-lens/blob/HEAD/transformers-4.32.0/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2407.01320","paper":"/paper/increasing-model-capacity-for-free-a-simple","title":"Increasing Model Capacity for Free: A Simple Strategy for Parameter Efficient Fine-tuning","date":"2024-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LINs-lab/CapaBoost","path":"src/transformers/activations_tf.py","file_url":"https://github.com/LINs-lab/CapaBoost/blob/HEAD/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2406.10960","paper":"/paper/escot-towards-interpretable-emotional-support","title":"ESCoT: Towards Interpretable Emotional Support Dialogue Systems","date":"2024-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-coai/Emotional-Support-Conversation","path":"codes/src/transformers/activations.py","file_url":"https://github.com/thu-coai/Emotional-Support-Conversation/blob/HEAD/codes/src/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2405.15179","paper":"/paper/vb-lora-extreme-parameter-efficient-fine","title":"VB-LoRA: Extreme Parameter Efficient Fine-Tuning with Vector Banks","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"leo-yangli/VB-LoRA","path":"NLU/NLU/src/transformers/activations.py","file_url":"https://github.com/leo-yangli/VB-LoRA/blob/HEAD/NLU/NLU/src/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2404.16367","paper":"/paper/learning-syntax-without-planting-trees","title":"Learning Syntax Without Planting Trees: Understanding When and Why Transformers Generalize Hierarchically","date":"2024-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kabirahuja2431/transformers-hg","path":"transformers/src/transformers/activations_tf.py","file_url":"https://github.com/kabirahuja2431/transformers-hg/blob/HEAD/transformers/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2403.10942","paper":"/paper/scantalk-3d-talking-heads-from-unregistered","title":"ScanTalk: 3D Talking Heads from Unregistered Scans","date":"2024-03-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"miccunifi/ScanTalk","path":"src/hubert/activations.py","file_url":"https://github.com/miccunifi/ScanTalk/blob/HEAD/src/hubert/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2402.18223","paper":"/paper/improving-open-ended-text-generation-via","title":"Improving Open-Ended Text Generation via Adaptive Decoding","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zwhong714/adaptive_decoding","path":"transformers-main/src/transformers/activations_tf.py","file_url":"https://github.com/zwhong714/adaptive_decoding/blob/HEAD/transformers-main/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2402.14789","paper":"/paper/self-guided-masked-autoencoders-for-domain","title":"Self-Guided Masked Autoencoders for Domain-Agnostic Self-Supervised Learning","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"johnathan-xie/sma","path":"src/transformers/activations_tf.py","file_url":"https://github.com/johnathan-xie/sma/blob/HEAD/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2402.02347","paper":"/paper/riemannian-preconditioned-lora-for-fine","title":"Riemannian Preconditioned LoRA for Fine-Tuning Foundation Models","date":"2024-02-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2401.10487","paper":"/paper/generative-dense-retrieval-memory-can-be-a","title":"Generative Dense Retrieval: Memory Can Be a Burden","date":"2024-01-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ypw0102/gdr","path":"GDR_model/transformers/activations.py","file_url":"https://github.com/ypw0102/gdr/blob/HEAD/GDR_model/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2312.12198","paper":"/paper/mask-grounding-for-referring-image","title":"Mask Grounding for Referring Image Segmentation","date":"2023-12-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yxchng/mask-grounding","path":"bert/activations.py","file_url":"https://github.com/yxchng/mask-grounding/blob/HEAD/bert/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"AGPL-3.0","inline_ok":false,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2311.07468","paper":"/paper/are-we-falling-in-a-middle-intelligence-trap","title":"An Analysis and Mitigation of the Reversal Curse","date":"2023-11-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"trestad/mitigating-reversal-curse","path":"transformers/src/transformers/activations_tf.py","file_url":"https://github.com/trestad/mitigating-reversal-curse/blob/HEAD/transformers/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2311.06761","paper":"/paper/learning-knowledge-enhanced-contextual","title":"Learning Knowledge-Enhanced Contextual Language Representations for Domain Natural Language Understanding","date":"2023-11-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba/EasyNLP","path":"easynlp/modelzoo/activations.py","file_url":"https://github.com/alibaba/EasyNLP/blob/HEAD/easynlp/modelzoo/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2310.02556","paper":"/paper/nola-networks-as-linear-combination-of-low","title":"NOLA: Compressing LoRA using Linear Combination of Random Basis","date":"2023-10-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"UCDvision/NOLA","path":"gpt/examples/NLG/src/model_nola.py","file_url":"https://github.com/UCDvision/NOLA/blob/HEAD/gpt/examples/NLG/src/model_nola.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2309.06363","paper":"/paper/2309-06363","title":"Learning to Predict Concept Ordering for Common Sense Generation","date":"2023-09-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tianhuizhang/concept_ordering","path":"bart/src/model/activations.py","file_url":"https://github.com/tianhuizhang/concept_ordering/blob/HEAD/bart/src/model/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2309.01017","paper":"/paper/contrastive-grouping-with-transformer-for-1","title":"Contrastive Grouping with Transformer for Referring Image Segmentation","date":"2023-09-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"toneyaya/cgformer","path":"bert/activations.py","file_url":"https://github.com/toneyaya/cgformer/blob/HEAD/bert/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2305.10614","paper":"/paper/token-wise-decomposition-of-autoregressive","title":"Token-wise Decomposition of Autoregressive Language Model Hidden States for Analyzing Model Predictions","date":"2023-05-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"byungdoh/llm_decomposition","path":"huggingface/src/transformers/activations_tf.py","file_url":"https://github.com/byungdoh/llm_decomposition/blob/HEAD/huggingface/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2301.12132","paper":"/paper/autopeft-automatic-configuration-search-for","title":"AutoPEFT: Automatic Configuration Search for Parameter-Efficient Fine-Tuning","date":"2023-01-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cambridgeltl/autopeft","path":"adapter-transformers-adapters3.1.0/src/transformers/activations_tf.py","file_url":"https://github.com/cambridgeltl/autopeft/blob/HEAD/adapter-transformers-adapters3.1.0/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2212.11185","paper":"/paper/entropy-and-distance-based-predictors-from","title":"Entropy- and Distance-Based Predictors From GPT-2 Attention Patterns Predict Reading Times Over and Above GPT-2 Surprisal","date":"2022-12-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"byungdoh/attn_dist","path":"huggingface/src/transformers/activations_tf.py","file_url":"https://github.com/byungdoh/attn_dist/blob/HEAD/huggingface/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2212.00921","paper":"/paper/agro-adversarial-discovery-of-error-prone","title":"AGRO: Adversarial Discovery of Error-prone groups for Robust Optimization","date":"2022-12-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bhargaviparanjape/robust-transformers","path":"src/transformers/activations_tf.py","file_url":"https://github.com/bhargaviparanjape/robust-transformers/blob/HEAD/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2211.05392","paper":"/paper/events-realm-event-reasoning-of-entity-states","title":"EvEntS ReaLM: Event Reasoning of Entity States via Language Models","date":"2022-11-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"spilioeve/eventsrealm","path":"transformers-single-all-attribute-prompt-experiments/src/transformers/activations_tf.py","file_url":"https://github.com/spilioeve/eventsrealm/blob/HEAD/transformers-single-all-attribute-prompt-experiments/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2202.00666","paper":"/paper/typical-decoding-for-natural-language","title":"Locally Typical Sampling","date":"2022-02-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cimeister/typical-sampling","path":"src/transformers/activations.py","file_url":"https://github.com/cimeister/typical-sampling/blob/HEAD/src/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2202.00666","paper":"/paper/typical-decoding-for-natural-language","title":"Locally Typical Sampling","date":"2022-02-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cimeister/typical-sampling","path":"src/transformers/activations_tf.py","file_url":"https://github.com/cimeister/typical-sampling/blob/HEAD/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2111.05498","paper":"/paper/attention-approximates-sparse-distributed","title":"Attention Approximates Sparse Distributed Memory","date":"2021-11-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"trentbrick/attention-approximates-sdm","path":"HugFace/src/transformers/activations.py","file_url":"https://github.com/trentbrick/attention-approximates-sdm/blob/HEAD/HugFace/src/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2111.00160","paper":"/paper/dsee-dually-sparsity-embedded-efficient-1","title":"DSEE: Dually Sparsity-embedded Efficient Tuning of Pre-trained Language Models","date":"2021-10-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vita-group/dsee","path":"non-GPT-2/src/transformers/activations.py","file_url":"https://github.com/vita-group/dsee/blob/HEAD/non-GPT-2/src/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2111.00160","paper":"/paper/dsee-dually-sparsity-embedded-efficient-1","title":"DSEE: Dually Sparsity-embedded Efficient Tuning of Pre-trained Language Models","date":"2021-10-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vita-group/dsee","path":"non-GPT-2/src/transformers/activations_tf.py","file_url":"https://github.com/vita-group/dsee/blob/HEAD/non-GPT-2/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2110.04366","paper":"/paper/towards-a-unified-view-of-parameter-efficient-1","title":"Towards a Unified View of Parameter-Efficient Transfer Learning","date":"2021-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jxhe/unify-parameter-efficient-tuning","path":"src/transformers/activations.py","file_url":"https://github.com/jxhe/unify-parameter-efficient-tuning/blob/HEAD/src/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2110.04366","paper":"/paper/towards-a-unified-view-of-parameter-efficient-1","title":"Towards a Unified View of Parameter-Efficient Transfer Learning","date":"2021-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jxhe/unify-parameter-efficient-tuning","path":"src/transformers/activations_tf.py","file_url":"https://github.com/jxhe/unify-parameter-efficient-tuning/blob/HEAD/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2109.03808","paper":"/paper/smelting-gold-and-silver-for-improved","title":"Smelting Gold and Silver for Improved Multilingual AMR-to-Text Generation","date":"2021-09-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"UKPLab/m-AMR2Text","path":"transformers/activations.py","file_url":"https://github.com/UKPLab/m-AMR2Text/blob/HEAD/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2109.03808","paper":"/paper/smelting-gold-and-silver-for-improved","title":"Smelting Gold and Silver for Improved Multilingual AMR-to-Text Generation","date":"2021-09-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"UKPLab/m-AMR2Text","path":"transformers/activations_tf.py","file_url":"https://github.com/UKPLab/m-AMR2Text/blob/HEAD/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ac1bc5cb23b3f2ec","mcp_get_code":{"code_sha256":"ac1bc5cb23b3f2ec"}},{"arxiv_id":"2107.00910","paper":"/paper/learned-token-pruning-for-transformers","title":"Learned Token Pruning for Transformers","date":"2021-07-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kssteven418/ltp","path":"src/transformers/activations.py","file_url":"https://github.com/kssteven418/ltp/blob/HEAD/src/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2107.00910","paper":"/paper/learned-token-pruning-for-transformers","title":"Learned Token Pruning for Transformers","date":"2021-07-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kssteven418/ltp","path":"src/transformers/activations_tf.py","file_url":"https://github.com/kssteven418/ltp/blob/HEAD/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"ac1bc5cb23b3f2ec","mcp_get_code":{"code_sha256":"ac1bc5cb23b3f2ec"}},{"arxiv_id":"2106.09248","paper":"/paper/x-fact-a-new-benchmark-dataset-for","title":"X-FACT: A New Benchmark Dataset for Multilingual Fact Checking","date":"2021-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"utahnlp/x-fact","path":"transformers/src/transformers/activations.py","file_url":"https://github.com/utahnlp/x-fact/blob/HEAD/transformers/src/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2106.00420","paper":"/paper/dialogue-oriented-pre-training","title":"Dialogue-oriented Pre-training","date":"2021-06-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xyease/Dialog-PrLM","path":"src/transformers/activations.py","file_url":"https://github.com/xyease/Dialog-PrLM/blob/HEAD/src/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2104.08400","paper":"/paper/structure-aware-abstractive-conversation","title":"Structure-Aware Abstractive Conversation Summarization via Discourse and Action Graphs","date":"2021-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"GT-SALT/Structure-Aware-BART","path":"transformers/src/transformers/activations.py","file_url":"https://github.com/GT-SALT/Structure-Aware-BART/blob/HEAD/transformers/src/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2104.08066","paper":"/paper/effect-of-vision-and-language-extensions-on","title":"Effect of Visual Extensions on Natural Language Understanding in Vision-and-Language Models","date":"2021-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alab-nii/eval_vl_glue","path":"eval_vl_glue/transformers_volta/activations.py","file_url":"https://github.com/alab-nii/eval_vl_glue/blob/HEAD/eval_vl_glue/transformers_volta/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2010.03957","paper":"/paper/transformers-for-modeling-physical-systems-1","title":"Transformers for Modeling Physical Systems","date":"2020-10-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zabaras/transformer-physx","path":"trphysx/transformer/utils.py","file_url":"https://github.com/zabaras/transformer-physx/blob/HEAD/trphysx/transformer/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b253e96edee87a52","mcp_get_code":{"code_sha256":"b253e96edee87a52"}},{"arxiv_id":"2007.07779","paper":"/paper/adapterhub-a-framework-for-adapting","title":"AdapterHub: A Framework for Adapting Transformers","date":"2020-07-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"adapter-hub/adapter-transformers-legacy","path":"src/transformers/activations_tf.py","file_url":"https://github.com/adapter-hub/adapter-transformers-legacy/blob/HEAD/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2024.findings-naacl.117","paper":null,"title":"arXiv:2024.findings-naacl.117","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"qqplot/dcpmi","path":"transformers/src/transformers/activations_tf.py","file_url":"https://github.com/qqplot/dcpmi/blob/HEAD/transformers/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}},{"arxiv_id":"2022.naacl-main.130","paper":null,"title":"arXiv:2022.naacl-main.130","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"parovicm/BADX","path":"src/transformers/activations.py","file_url":"https://github.com/parovicm/BADX/blob/HEAD/src/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a4475703ff58ecf9","mcp_get_code":{"code_sha256":"a4475703ff58ecf9"}},{"arxiv_id":"2022.naacl-main.130","paper":null,"title":"arXiv:2022.naacl-main.130","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"parovicm/BADX","path":"src/transformers/activations_tf.py","file_url":"https://github.com/parovicm/BADX/blob/HEAD/src/transformers/activations_tf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"37a5eed2dbd663ca","mcp_get_code":{"code_sha256":"37a5eed2dbd663ca"}}]}