{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/gelu-new","entry":"gelu_new","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":55,"n_papers_ran":49,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":14,"n_samples_ran":9,"n_samples_fingerprinted":9,"n_places":61,"n_places_pointer_only":18,"by_status":{"ran_honours":4,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":4,"unverified":5},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2502.07707","paper":"/paper/prvql-progressive-knowledge-guided-refinement","title":"PRVQL: Progressive Knowledge-guided Refinement for Robust Egocentric Visual Query Localization","date":"2025-02-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fb-reps/PRVQL","path":"bert_model/bert_module.py","file_url":"https://github.com/fb-reps/PRVQL/blob/HEAD/bert_model/bert_module.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2410.05748","paper":"/paper/label-confidence-weighted-learning-for-target","title":"Label Confidence Weighted Learning for Target-level Sentence Simplification","date":"2024-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"astro-jon/LCWL","path":"src/transformers/activations.py","file_url":"https://github.com/astro-jon/LCWL/blob/HEAD/src/transformers/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"45bb87451230d5e8","mcp_get_code":{"code_sha256":"45bb87451230d5e8"}},{"arxiv_id":"2406.10960","paper":"/paper/escot-towards-interpretable-emotional-support","title":"ESCoT: Towards Interpretable Emotional Support Dialogue Systems","date":"2024-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-coai/Emotional-Support-Conversation","path":"codes/src/transformers/activations.py","file_url":"https://github.com/thu-coai/Emotional-Support-Conversation/blob/HEAD/codes/src/transformers/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"45bb87451230d5e8","mcp_get_code":{"code_sha256":"45bb87451230d5e8"}},{"arxiv_id":"2406.10960","paper":"/paper/escot-towards-interpretable-emotional-support","title":"ESCoT: Towards Interpretable Emotional Support Dialogue Systems","date":"2024-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-coai/Emotional-Support-Conversation","path":"codes/src/transformers/activations_tf.py","file_url":"https://github.com/thu-coai/Emotional-Support-Conversation/blob/HEAD/codes/src/transformers/activations_tf.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"b01a23b01774f91e","mcp_get_code":{"code_sha256":"b01a23b01774f91e"}},{"arxiv_id":"2405.15179","paper":"/paper/vb-lora-extreme-parameter-efficient-fine","title":"VB-LoRA: Extreme Parameter Efficient Fine-Tuning with Vector Banks","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"leo-yangli/VB-LoRA","path":"NLU/NLU/src/transformers/activations.py","file_url":"https://github.com/leo-yangli/VB-LoRA/blob/HEAD/NLU/NLU/src/transformers/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"45bb87451230d5e8","mcp_get_code":{"code_sha256":"45bb87451230d5e8"}},{"arxiv_id":"2404.19563","paper":"/paper/repeval-effective-text-evaluation-with-llm","title":"RepEval: Effective Text Evaluation with LLM Representation","date":"2024-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shikib/usr","path":"transformers/modeling_bert.py","file_url":"https://github.com/shikib/usr/blob/HEAD/transformers/modeling_bert.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2403.10942","paper":"/paper/scantalk-3d-talking-heads-from-unregistered","title":"ScanTalk: 3D Talking Heads from Unregistered Scans","date":"2024-03-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"miccunifi/ScanTalk","path":"src/hubert/activations.py","file_url":"https://github.com/miccunifi/ScanTalk/blob/HEAD/src/hubert/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"45bb87451230d5e8","mcp_get_code":{"code_sha256":"45bb87451230d5e8"}},{"arxiv_id":"2402.02347","paper":"/paper/riemannian-preconditioned-lora-for-fine","title":"Riemannian Preconditioned LoRA for Fine-Tuning Foundation Models","date":"2024-02-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"dc9ffc0f29e7fa3d","mcp_get_code":{"code_sha256":"dc9ffc0f29e7fa3d"}},{"arxiv_id":"2401.10487","paper":"/paper/generative-dense-retrieval-memory-can-be-a","title":"Generative Dense Retrieval: Memory Can Be a Burden","date":"2024-01-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ypw0102/gdr","path":"GDR_model/transformers/activations.py","file_url":"https://github.com/ypw0102/gdr/blob/HEAD/GDR_model/transformers/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"aef07bb1e2787431","mcp_get_code":{"code_sha256":"aef07bb1e2787431"}},{"arxiv_id":"2401.10487","paper":"/paper/generative-dense-retrieval-memory-can-be-a","title":"Generative Dense Retrieval: Memory Can Be a Burden","date":"2024-01-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ypw0102/gdr","path":"GDR_model/transformers/activations_tf.py","file_url":"https://github.com/ypw0102/gdr/blob/HEAD/GDR_model/transformers/activations_tf.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"13718ddda6b66f8e","mcp_get_code":{"code_sha256":"13718ddda6b66f8e"}},{"arxiv_id":"2312.12198","paper":"/paper/mask-grounding-for-referring-image","title":"Mask Grounding for Referring Image Segmentation","date":"2023-12-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yxchng/mask-grounding","path":"bert/activations.py","file_url":"https://github.com/yxchng/mask-grounding/blob/HEAD/bert/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"AGPL-3.0","inline_ok":false,"code_sha256_prefix":"dc9ffc0f29e7fa3d","mcp_get_code":{"code_sha256":"dc9ffc0f29e7fa3d"}},{"arxiv_id":"2311.06761","paper":"/paper/learning-knowledge-enhanced-contextual","title":"Learning Knowledge-Enhanced Contextual Language Representations for Domain Natural Language Understanding","date":"2023-11-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba/EasyNLP","path":"easynlp/modelzoo/activations.py","file_url":"https://github.com/alibaba/EasyNLP/blob/HEAD/easynlp/modelzoo/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"45bb87451230d5e8","mcp_get_code":{"code_sha256":"45bb87451230d5e8"}},{"arxiv_id":"2310.02556","paper":"/paper/nola-networks-as-linear-combination-of-low","title":"NOLA: Compressing LoRA using Linear Combination of Random Basis","date":"2023-10-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"UCDvision/NOLA","path":"gpt/examples/NLG/src/model_nola.py","file_url":"https://github.com/UCDvision/NOLA/blob/HEAD/gpt/examples/NLG/src/model_nola.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dc9ffc0f29e7fa3d","mcp_get_code":{"code_sha256":"dc9ffc0f29e7fa3d"}},{"arxiv_id":"2309.06363","paper":"/paper/2309-06363","title":"Learning to Predict Concept Ordering for Common Sense Generation","date":"2023-09-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tianhuizhang/concept_ordering","path":"bart/src/model/activations.py","file_url":"https://github.com/tianhuizhang/concept_ordering/blob/HEAD/bart/src/model/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"dc9ffc0f29e7fa3d","mcp_get_code":{"code_sha256":"dc9ffc0f29e7fa3d"}},{"arxiv_id":"2309.01017","paper":"/paper/contrastive-grouping-with-transformer-for-1","title":"Contrastive Grouping with Transformer for Referring Image Segmentation","date":"2023-09-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"toneyaya/cgformer","path":"bert/activations.py","file_url":"https://github.com/toneyaya/cgformer/blob/HEAD/bert/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dc9ffc0f29e7fa3d","mcp_get_code":{"code_sha256":"dc9ffc0f29e7fa3d"}},{"arxiv_id":"2306.15006","paper":"/paper/dnabert-2-efficient-foundation-model-and","title":"DNABERT-2: Efficient Foundation Model and Benchmark For Multi-Species Genome","date":"2023-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jerryji1993/dnabert","path":"src/transformers/activations.py","file_url":"https://github.com/jerryji1993/dnabert/blob/HEAD/src/transformers/activations.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2306.08891","paper":"/paper/interleaving-pre-trained-language-models-and","title":"Interleaving Pre-Trained Language Models and Large Language Models for Zero-Shot NL2SQL Generation","date":"2023-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ruc-datalab/zeronl2sql","path":"src/utils/aligner.py","file_url":"https://github.com/ruc-datalab/zeronl2sql/blob/HEAD/src/utils/aligner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"227f586d79c9675d","mcp_get_code":{"code_sha256":"227f586d79c9675d"}},{"arxiv_id":"2212.03506","paper":"/paper/wider-closer-mixture-of-short-channel","title":"WIDER & CLOSER: Mixture of Short-channel Distillers for Zero-shot Cross-lingual Named Entity Recognition","date":"2022-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mckysse/msd","path":"transformers/modeling_bert.py","file_url":"https://github.com/mckysse/msd/blob/HEAD/transformers/modeling_bert.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2212.03506","paper":"/paper/wider-closer-mixture-of-short-channel","title":"WIDER & CLOSER: Mixture of Short-channel Distillers for Zero-shot Cross-lingual Named Entity Recognition","date":"2022-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mckysse/msd","path":"transformers/modeling_tf_bert.py","file_url":"https://github.com/mckysse/msd/blob/HEAD/transformers/modeling_tf_bert.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"83257a2b015273fa","mcp_get_code":{"code_sha256":"83257a2b015273fa"}},{"arxiv_id":"2211.02816","paper":"/paper/pasta-table-operations-aware-fact","title":"PASTA: Table-Operations Aware Fact Verification via Sentence-Table Cloze Pre-training","date":"2022-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ruc-datalab/PASTA","path":"src/utils/pasta_mlm_model.py","file_url":"https://github.com/ruc-datalab/PASTA/blob/HEAD/src/utils/pasta_mlm_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"283f4bd4ad4cc1ed","mcp_get_code":{"code_sha256":"283f4bd4ad4cc1ed"}},{"arxiv_id":"2211.01335","paper":"/paper/chinese-clip-contrastive-vision-language","title":"Chinese CLIP: Contrastive Vision-Language Pretraining in Chinese","date":"2022-11-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ofa-sys/chinese-clip","path":"cn_clip/clip/modeling_bert.py","file_url":"https://github.com/ofa-sys/chinese-clip/blob/HEAD/cn_clip/clip/modeling_bert.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2210.08714","paper":"/paper/selective-query-guided-debiasing-network-for","title":"Selective Query-guided Debiasing for Video Corpus Moment Retrieval","date":"2022-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dbstjswo505/SQuiDNet","path":"model/squidnet.py","file_url":"https://github.com/dbstjswo505/SQuiDNet/blob/HEAD/model/squidnet.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"de48fec0ec9764b0","mcp_get_code":{"code_sha256":"de48fec0ec9764b0"}},{"arxiv_id":"2203.11431","paper":"/paper/task-guided-disentangled-tuning-for","title":"Task-guided Disentangled Tuning for Pretrained Language Models","date":"2022-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lemon0830/TDT","path":"TDT/transformers/modeling_bert.py","file_url":"https://github.com/lemon0830/TDT/blob/HEAD/TDT/transformers/modeling_bert.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2202.00666","paper":"/paper/typical-decoding-for-natural-language","title":"Locally Typical Sampling","date":"2022-02-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cimeister/typical-sampling","path":"src/transformers/activations.py","file_url":"https://github.com/cimeister/typical-sampling/blob/HEAD/src/transformers/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"45bb87451230d5e8","mcp_get_code":{"code_sha256":"45bb87451230d5e8"}},{"arxiv_id":"2111.05498","paper":"/paper/attention-approximates-sparse-distributed","title":"Attention Approximates Sparse Distributed Memory","date":"2021-11-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"trentbrick/attention-approximates-sdm","path":"HugFace/src/transformers/activations.py","file_url":"https://github.com/trentbrick/attention-approximates-sdm/blob/HEAD/HugFace/src/transformers/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"45bb87451230d5e8","mcp_get_code":{"code_sha256":"45bb87451230d5e8"}},{"arxiv_id":"2111.00160","paper":"/paper/dsee-dually-sparsity-embedded-efficient-1","title":"DSEE: Dually Sparsity-embedded Efficient Tuning of Pre-trained Language Models","date":"2021-10-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vita-group/dsee","path":"non-GPT-2/src/transformers/activations.py","file_url":"https://github.com/vita-group/dsee/blob/HEAD/non-GPT-2/src/transformers/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"45bb87451230d5e8","mcp_get_code":{"code_sha256":"45bb87451230d5e8"}},{"arxiv_id":"2110.04366","paper":"/paper/towards-a-unified-view-of-parameter-efficient-1","title":"Towards a Unified View of Parameter-Efficient Transfer Learning","date":"2021-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jxhe/unify-parameter-efficient-tuning","path":"src/transformers/activations.py","file_url":"https://github.com/jxhe/unify-parameter-efficient-tuning/blob/HEAD/src/transformers/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"45bb87451230d5e8","mcp_get_code":{"code_sha256":"45bb87451230d5e8"}},{"arxiv_id":"2109.06067","paper":"/paper/pack-together-entity-and-relation-extraction","title":"Packed Levitated Marker for Entity and Relation Extraction","date":"2021-09-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp/pl-marker","path":"transformers/src/transformers/activations.py","file_url":"https://github.com/thunlp/pl-marker/blob/HEAD/transformers/src/transformers/activations.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2109.03808","paper":"/paper/smelting-gold-and-silver-for-improved","title":"Smelting Gold and Silver for Improved Multilingual AMR-to-Text Generation","date":"2021-09-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"UKPLab/m-AMR2Text","path":"transformers/activations.py","file_url":"https://github.com/UKPLab/m-AMR2Text/blob/HEAD/transformers/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"45bb87451230d5e8","mcp_get_code":{"code_sha256":"45bb87451230d5e8"}},{"arxiv_id":"2107.00910","paper":"/paper/learned-token-pruning-for-transformers","title":"Learned Token Pruning for Transformers","date":"2021-07-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kssteven418/ltp","path":"src/transformers/activations.py","file_url":"https://github.com/kssteven418/ltp/blob/HEAD/src/transformers/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"45bb87451230d5e8","mcp_get_code":{"code_sha256":"45bb87451230d5e8"}},{"arxiv_id":"2106.11310","paper":"/paper/towards-long-form-video-understanding-1","title":"Towards Long-Form Video Understanding","date":"2021-06-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chaoyuaw/lvu","path":"src/models/modeling_bert.py","file_url":"https://github.com/chaoyuaw/lvu/blob/HEAD/src/models/modeling_bert.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2106.09248","paper":"/paper/x-fact-a-new-benchmark-dataset-for","title":"X-FACT: A New Benchmark Dataset for Multilingual Fact Checking","date":"2021-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"utahnlp/x-fact","path":"transformers/src/transformers/activations.py","file_url":"https://github.com/utahnlp/x-fact/blob/HEAD/transformers/src/transformers/activations.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dc9ffc0f29e7fa3d","mcp_get_code":{"code_sha256":"dc9ffc0f29e7fa3d"}},{"arxiv_id":"2106.04632","paper":"/paper/value-a-multi-task-benchmark-for-video-and","title":"VALUE: A Multi-Task Benchmark for Video-and-Language Understanding Evaluation","date":"2021-06-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"VALUE-Leaderboard/StarterCode","path":"model/layers.py","file_url":"https://github.com/VALUE-Leaderboard/StarterCode/blob/HEAD/model/layers.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"de48fec0ec9764b0","mcp_get_code":{"code_sha256":"de48fec0ec9764b0"}},{"arxiv_id":"2106.00420","paper":"/paper/dialogue-oriented-pre-training","title":"Dialogue-oriented Pre-training","date":"2021-06-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xyease/Dialog-PrLM","path":"src/transformers/activations.py","file_url":"https://github.com/xyease/Dialog-PrLM/blob/HEAD/src/transformers/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"aef07bb1e2787431","mcp_get_code":{"code_sha256":"aef07bb1e2787431"}},{"arxiv_id":"2106.00420","paper":"/paper/dialogue-oriented-pre-training","title":"Dialogue-oriented Pre-training","date":"2021-06-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xyease/Dialog-PrLM","path":"src/transformers/activations_tf.py","file_url":"https://github.com/xyease/Dialog-PrLM/blob/HEAD/src/transformers/activations_tf.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"13718ddda6b66f8e","mcp_get_code":{"code_sha256":"13718ddda6b66f8e"}},{"arxiv_id":"2105.12002","paper":"/paper/super-tickets-in-pre-trained-language-models","title":"Super Tickets in Pre-Trained Language Models: From Model Compression to Improving Generalization","date":"2021-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cliang1453/super-structured-lottery-tickets","path":"module/modeling_bert.py","file_url":"https://github.com/cliang1453/super-structured-lottery-tickets/blob/HEAD/module/modeling_bert.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2104.08400","paper":"/paper/structure-aware-abstractive-conversation","title":"Structure-Aware Abstractive Conversation Summarization via Discourse and Action Graphs","date":"2021-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"GT-SALT/Structure-Aware-BART","path":"transformers/src/transformers/activations.py","file_url":"https://github.com/GT-SALT/Structure-Aware-BART/blob/HEAD/transformers/src/transformers/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"aef07bb1e2787431","mcp_get_code":{"code_sha256":"aef07bb1e2787431"}},{"arxiv_id":"2104.08400","paper":"/paper/structure-aware-abstractive-conversation","title":"Structure-Aware Abstractive Conversation Summarization via Discourse and Action Graphs","date":"2021-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"GT-SALT/Structure-Aware-BART","path":"transformers/src/transformers/activations_tf.py","file_url":"https://github.com/GT-SALT/Structure-Aware-BART/blob/HEAD/transformers/src/transformers/activations_tf.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"13718ddda6b66f8e","mcp_get_code":{"code_sha256":"13718ddda6b66f8e"}},{"arxiv_id":"2104.08066","paper":"/paper/effect-of-vision-and-language-extensions-on","title":"Effect of Visual Extensions on Natural Language Understanding in Vision-and-Language Models","date":"2021-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alab-nii/eval_vl_glue","path":"eval_vl_glue/transformers_volta/activations.py","file_url":"https://github.com/alab-nii/eval_vl_glue/blob/HEAD/eval_vl_glue/transformers_volta/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"45bb87451230d5e8","mcp_get_code":{"code_sha256":"45bb87451230d5e8"}},{"arxiv_id":"2103.16110","paper":"/paper/kaleido-bert-vision-language-pre-training-on","title":"Kaleido-BERT: Vision-Language Pre-training on Fashion Domain","date":"2021-03-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mczhuge/Kaleido-BERT","path":"easytransfer/layers/activations.py","file_url":"https://github.com/mczhuge/Kaleido-BERT/blob/HEAD/easytransfer/layers/activations.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"83257a2b015273fa","mcp_get_code":{"code_sha256":"83257a2b015273fa"}},{"arxiv_id":"2011.01513","paper":"/paper/charbert-character-aware-pre-trained-language","title":"CharBERT: Character-aware Pre-trained Language Model","date":"2020-11-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wtma/CharBERT","path":"modeling/modeling_bert.py","file_url":"https://github.com/wtma/CharBERT/blob/HEAD/modeling/modeling_bert.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2010.10392","paper":"/paper/characterbert-reconciling-elmo-and-bert-for","title":"CharacterBERT: Reconciling ELMo and BERT for Word-Level Open-Vocabulary Representations From Characters","date":"2020-10-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"IMPLabUniPr/UniParma-at-semeval-2021-task-5","path":"transformers/modeling_bert.py","file_url":"https://github.com/IMPLabUniPr/UniParma-at-semeval-2021-task-5/blob/HEAD/transformers/modeling_bert.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2010.03957","paper":"/paper/transformers-for-modeling-physical-systems-1","title":"Transformers for Modeling Physical Systems","date":"2020-10-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zabaras/transformer-physx","path":"trphysx/transformer/utils.py","file_url":"https://github.com/zabaras/transformer-physx/blob/HEAD/trphysx/transformer/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"32d5035f6c9444f0","mcp_get_code":{"code_sha256":"32d5035f6c9444f0"}},{"arxiv_id":"2009.12719","paper":"/paper/stylized-dialogue-response-generation-using","title":"Stylized Dialogue Response Generation Using Stylized Unpaired Texts","date":"2020-09-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"silverriver/Stylized_Dialog","path":"TCFC/bt_beam/model/activations.py","file_url":"https://github.com/silverriver/Stylized_Dialog/blob/HEAD/TCFC/bt_beam/model/activations.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2009.05166","paper":"/paper/filter-an-enhanced-fusion-method-for-cross","title":"FILTER: An Enhanced Fusion Method for Cross-lingual Language Understanding","date":"2020-09-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuwfan/FILTER","path":"src/transformers/activations.py","file_url":"https://github.com/yuwfan/FILTER/blob/HEAD/src/transformers/activations.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2007.12223","paper":"/paper/the-lottery-ticket-hypothesis-for-pre-trained","title":"The Lottery Ticket Hypothesis for Pre-trained BERT Networks","date":"2020-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TAMU-VITA/BERT-Tickets","path":"transformers-master/src/transformers/modeling_tf_bert.py","file_url":"https://github.com/TAMU-VITA/BERT-Tickets/blob/HEAD/transformers-master/src/transformers/modeling_tf_bert.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"83257a2b015273fa","mcp_get_code":{"code_sha256":"83257a2b015273fa"}},{"arxiv_id":"2006.04884","paper":"/paper/on-the-stability-of-fine-tuning-bert","title":"On the Stability of Fine-tuning BERT: Misconceptions, Explanations, and Strong Baselines","date":"2020-06-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uds-lsv/bert-stable-fine-tuning","path":"src/transformers/activations.py","file_url":"https://github.com/uds-lsv/bert-stable-fine-tuning/blob/HEAD/src/transformers/activations.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2006.04884","paper":"/paper/on-the-stability-of-fine-tuning-bert","title":"On the Stability of Fine-tuning BERT: Misconceptions, Explanations, and Strong Baselines","date":"2020-06-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uds-lsv/bert-stable-fine-tuning","path":"src/transformers/modeling_tf_bert.py","file_url":"https://github.com/uds-lsv/bert-stable-fine-tuning/blob/HEAD/src/transformers/modeling_tf_bert.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"83257a2b015273fa","mcp_get_code":{"code_sha256":"83257a2b015273fa"}},{"arxiv_id":"2006.04558","paper":"/paper/fastspeech-2-fast-and-high-quality-end-to-end","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","date":"2020-06-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dathudeptrai/TensorflowTTS","path":"tensorflow_tts/models/fastspeech2.py","file_url":"https://github.com/dathudeptrai/TensorflowTTS/blob/HEAD/tensorflow_tts/models/fastspeech2.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f6c9754063b07733","mcp_get_code":{"code_sha256":"f6c9754063b07733"}},{"arxiv_id":"2006.03535","paper":"/paper/cocon-a-self-supervised-approach-for","title":"CoCon: A Self-Supervised Approach for Controlled Text Generation","date":"2020-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2005.00770","paper":"/paper/exploring-and-predicting-transferability","title":"Exploring and Predicting Transferability across NLP Tasks","date":"2020-05-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tuvuumass/task-transferability","path":"transformers/modeling_task_embeddings.py","file_url":"https://github.com/tuvuumass/task-transferability/blob/HEAD/transformers/modeling_task_embeddings.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2005.00200","paper":"/paper/hero-hierarchical-encoder-for-video-language","title":"HERO: Hierarchical Encoder for Video+Language Omni-representation Pre-training","date":"2020-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"linjieli222/HERO","path":"model/layers.py","file_url":"https://github.com/linjieli222/HERO/blob/HEAD/model/layers.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"de48fec0ec9764b0","mcp_get_code":{"code_sha256":"de48fec0ec9764b0"}},{"arxiv_id":"2004.10964","paper":"/paper/don-t-stop-pretraining-adapt-language-models","title":"Don't Stop Pretraining: Adapt Language Models to Domains and Tasks","date":"2020-04-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shizhediao/t-dna","path":"TDNA/util.py","file_url":"https://github.com/shizhediao/t-dna/blob/HEAD/TDNA/util.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"aef07bb1e2787431","mcp_get_code":{"code_sha256":"aef07bb1e2787431"}},{"arxiv_id":"2004.03844","paper":"/paper/poor-man-s-bert-smaller-and-faster","title":"On the Effect of Dropping Layers of Pre-trained Transformer Models","date":"2020-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hsajjad/transformers","path":"src/transformers/activations.py","file_url":"https://github.com/hsajjad/transformers/blob/HEAD/src/transformers/activations.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2002.10345","paper":"/paper/improving-bert-fine-tuning-via-self-ensemble","title":"Improving BERT Fine-Tuning via Self-Ensemble and Self-Distillation","date":"2020-02-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lonePatient/BERT-SDA","path":"models/transformers/modeling_bert.py","file_url":"https://github.com/lonePatient/BERT-SDA/blob/HEAD/models/transformers/modeling_bert.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"1911.03631","paper":"/paper/hierarchical-graph-network-for-multi-hop","title":"Hierarchical Graph Network for Multi-hop Question Answering","date":"2019-11-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuwfan/HGN","path":"transformers/modeling_bert.py","file_url":"https://github.com/yuwfan/HGN/blob/HEAD/transformers/modeling_bert.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"1910.01108","paper":"/paper/distilbert-a-distilled-version-of-bert","title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","date":"2019-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mkavim/finetune_bert","path":"finetune/modeling_distilbert.py","file_url":"https://github.com/mkavim/finetune_bert/blob/HEAD/finetune/modeling_distilbert.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5b3fd00218623502","mcp_get_code":{"code_sha256":"5b3fd00218623502"}},{"arxiv_id":"1909.11942","paper":"/paper/albert-a-lite-bert-for-self-supervised","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","date":"2019-09-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Soikonomou/albert_final","path":"src/model/ALBERT/modeling_bert.py","file_url":"https://github.com/Soikonomou/albert_final/blob/HEAD/src/model/ALBERT/modeling_bert.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lonePatient/Bert-Multi-Label-Text-Classification","path":"pybert/model/albert/modeling_bert.py","file_url":"https://github.com/lonePatient/Bert-Multi-Label-Text-Classification/blob/HEAD/pybert/model/albert/modeling_bert.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"77601724cad03f95","mcp_get_code":{"code_sha256":"77601724cad03f95"}},{"arxiv_id":"2025.findings-emnlp.70","paper":null,"title":"arXiv:2025.findings-emnlp.70","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"passing2961/EmpGPT-3","path":"from_epitome/activations.py","file_url":"https://github.com/passing2961/EmpGPT-3/blob/HEAD/from_epitome/activations.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4a4442fe138425f8","mcp_get_code":{"code_sha256":"4a4442fe138425f8"}},{"arxiv_id":"2022.naacl-main.130","paper":null,"title":"arXiv:2022.naacl-main.130","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"parovicm/BADX","path":"src/transformers/activations.py","file_url":"https://github.com/parovicm/BADX/blob/HEAD/src/transformers/activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"45bb87451230d5e8","mcp_get_code":{"code_sha256":"45bb87451230d5e8"}}]}