{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/load-tf-weights-in-gpt2","entry":"load_tf_weights_in_gpt2","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":68,"n_papers_ran":0,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":14,"n_samples_ran":0,"n_samples_fingerprinted":0,"n_places":69,"n_places_pointer_only":14,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":0,"unverified":14},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2501.13883","paper":"/paper/utilizing-evolution-strategies-to-train","title":"Utilizing Evolution Strategies to Train Transformers in Reinforcement Learning","date":"2025-01-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mafi412/evolution-strategies-and-decision-transformers","path":"codebase/components/decision_transformer/gym/models/trajectory_gpt2.py","file_url":"https://github.com/mafi412/evolution-strategies-and-decision-transformers/blob/HEAD/codebase/components/decision_transformer/gym/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2412.04445","paper":"/paper/moto-latent-motion-token-as-the-bridging","title":"Moto: Latent Motion Token as the Bridging Language for Learning Robot Manipulation from Videos","date":"2024-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tencentarc/moto","path":"moto_gpt/src/models/trajectory_gpt2.py","file_url":"https://github.com/tencentarc/moto/blob/HEAD/moto_gpt/src/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2411.11364","paper":"/paper/continual-task-learning-through-adaptive","title":"Continual Task Learning through Adaptive Policy Self-Composition","date":"2024-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"charleshsc/CompoFormer","path":"dt/trajectory_gpt2.py","file_url":"https://github.com/charleshsc/CompoFormer/blob/HEAD/dt/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2411.02359","paper":"/paper/deer-vla-dynamic-inference-of-multimodal","title":"DeeR-VLA: Dynamic Inference of Multimodal Large Language Models for Efficient Robot Execution","date":"2024-11-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yueyang130/DeeR-VLA","path":"robot_flamingo/models/trajectory_gpt2.py","file_url":"https://github.com/yueyang130/DeeR-VLA/blob/HEAD/robot_flamingo/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2411.02280","paper":"/paper/the-llm-language-network-a-neuroscientific","title":"The LLM Language Network: A Neuroscientific Approach for Identifying Causally Task-Relevant Units","date":"2024-11-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bkhmsi/llm-localization","path":"models/modeling_gpt2.py","file_url":"https://github.com/bkhmsi/llm-localization/blob/HEAD/models/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"eae14ebac1cd9c52","mcp_get_code":{"code_sha256":"eae14ebac1cd9c52"}},{"arxiv_id":"2411.01146","paper":"/paper/task-aware-harmony-multi-task-decision","title":"Task-Aware Harmony Multi-Task Decision Transformer for Offline Reinforcement Learning","date":"2024-11-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"charleshsc/HarmoDT","path":"prompt_dt/trajectory_gpt2.py","file_url":"https://github.com/charleshsc/HarmoDT/blob/HEAD/prompt_dt/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2410.24218","paper":"/paper/teaching-embodied-reinforcement-learning","title":"Teaching Embodied Reinforcement Learning Agents: Informativeness and Diversity of Language Use","date":"2024-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sled-group/teachable_rl","path":"alfworld/decision_transformer/models/trajectory_gpt2.py","file_url":"https://github.com/sled-group/teachable_rl/blob/HEAD/alfworld/decision_transformer/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2410.24108","paper":"/paper/reinforcement-learning-gradients-as-vitamin","title":"Reinforcement Learning Gradients as Vitamin for Online Finetuning Decision Transformers","date":"2024-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kaiyan289/rl_as_vitamin_for_online_decision_transformers","path":"decision_transformer/models/trajectory_gpt2.py","file_url":"https://github.com/kaiyan289/rl_as_vitamin_for_online_decision_transformers/blob/HEAD/decision_transformer/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2410.17897","paper":"/paper/value-residual-learning-for-alleviating","title":"Value Residual Learning For Alleviating Attention Concentration In Transformers","date":"2024-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Zcchill/Value-Residual-Learning","path":"src/modeling/modeling_gpt2_baseline.py","file_url":"https://github.com/Zcchill/Value-Residual-Learning/blob/HEAD/src/modeling/modeling_gpt2_baseline.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"eae14ebac1cd9c52","mcp_get_code":{"code_sha256":"eae14ebac1cd9c52"}},{"arxiv_id":"2410.11448","paper":"/paper/meta-dt-offline-meta-rl-as-conditional","title":"Meta-DT: Offline Meta-RL as Conditional Sequence Modeling with World Model Disentanglement","date":"2024-10-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"NJU-RL/Meta-DT","path":"meta_dt/trajectory_gpt2.py","file_url":"https://github.com/NJU-RL/Meta-DT/blob/HEAD/meta_dt/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2409.01193","paper":"/paper/clibe-detecting-dynamic-backdoors-in","title":"CLIBE: Detecting Dynamic Backdoors in Transformer-based NLP Models","date":"2024-09-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"raytsang123/clibe","path":"generative_backdoors/detection/modeling_gpt2.py","file_url":"https://github.com/raytsang123/clibe/blob/HEAD/generative_backdoors/detection/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3665fea6427d655d","mcp_get_code":{"code_sha256":"3665fea6427d655d"}},{"arxiv_id":"2408.14368","paper":"/paper/gr-mg-leveraging-partially-annotated-data-via","title":"GR-MG: Leveraging Partially Annotated Data via Multi-Modal Goal-Conditioned Policy","date":"2024-08-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/GR-MG","path":"policy/model/gpt2.py","file_url":"https://github.com/bytedance/GR-MG/blob/HEAD/policy/model/gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3665fea6427d655d","mcp_get_code":{"code_sha256":"3665fea6427d655d"}},{"arxiv_id":"2407.18414","paper":"/paper/adversarial-robust-decision-transformer","title":"Adversarially Robust Decision Transformer","date":"2024-07-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xiaohangt/ardt","path":"decision_transformer/decision_transformer/models/trajectory_gpt2.py","file_url":"https://github.com/xiaohangt/ardt/blob/HEAD/decision_transformer/decision_transformer/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bc9fb74e483fb901","mcp_get_code":{"code_sha256":"bc9fb74e483fb901"}},{"arxiv_id":"2407.16920","paper":"/paper/train-attention-meta-learning-where-to-focus","title":"Train-Attention: Meta-Learning Where to Focus in Continual Knowledge Learning","date":"2024-07-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ybseo-academy/TAALM","path":"utils/kadapter/Llama2_Model_Kadapter.py","file_url":"https://github.com/ybseo-academy/TAALM/blob/HEAD/utils/kadapter/Llama2_Model_Kadapter.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cb4e99dd4b50f69d","mcp_get_code":{"code_sha256":"cb4e99dd4b50f69d"}},{"arxiv_id":"2406.12382","paper":"/paper/from-instance-training-to-instruction","title":"From Instance Training to Instruction Learning: Task Adapters Generation from Instructions","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Xnhyacinth/TAGI","path":"src/modeling_gpt2.py","file_url":"https://github.com/Xnhyacinth/TAGI/blob/HEAD/src/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bc9fb74e483fb901","mcp_get_code":{"code_sha256":"bc9fb74e483fb901"}},{"arxiv_id":"2406.01917","paper":"/paper/gomaa-geo-goal-modality-agnostic-active-geo","title":"GOMAA-Geo: GOal Modality Agnostic Active Geo-localization","date":"2024-06-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mvrl/gomaa-geo","path":"models/model_gpt.py","file_url":"https://github.com/mvrl/gomaa-geo/blob/HEAD/models/model_gpt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3665fea6427d655d","mcp_get_code":{"code_sha256":"3665fea6427d655d"}},{"arxiv_id":"2405.17098","paper":"/paper/q-value-regularized-transformer-for-offline","title":"Q-value Regularized Transformer for Offline Reinforcement Learning","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"charleshsc/qt","path":"decision_transformer/models/trajectory_gpt2.py","file_url":"https://github.com/charleshsc/qt/blob/HEAD/decision_transformer/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2404.19563","paper":"/paper/repeval-effective-text-evaluation-with-llm","title":"RepEval: Effective Text Evaluation with LLM Representation","date":"2024-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shikib/usr","path":"transformers/modeling_gpt2.py","file_url":"https://github.com/shikib/usr/blob/HEAD/transformers/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"654e91efc679fe84","mcp_get_code":{"code_sha256":"654e91efc679fe84"}},{"arxiv_id":"2404.10464","paper":"/paper/destein-navigating-detoxification-of-language","title":"DESTEIN: Navigating Detoxification of Language Models via Universal Steering Pairs and Head-wise Activation Fusion","date":"2024-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lizlizli/destein","path":"modeling/modeling_gpt2.py","file_url":"https://github.com/lizlizli/destein/blob/HEAD/modeling/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3665fea6427d655d","mcp_get_code":{"code_sha256":"3665fea6427d655d"}},{"arxiv_id":"2403.19925","paper":"/paper/decision-mamba-reinforcement-learning-via","title":"Decision Mamba: Reinforcement Learning via Sequence Modeling with Selective State Spaces","date":"2024-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"toshihiro-ota/decision-mamba","path":"gym/models/trajectory_gpt2.py","file_url":"https://github.com/toshihiro-ota/decision-mamba/blob/HEAD/gym/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2403.09054","paper":"/paper/keyformer-kv-cache-reduction-through-key","title":"Keyformer: KV Cache Reduction through Key Tokens Selection for Efficient Generative Inference","date":"2024-03-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"d-matrix-ai/keyformer-llm","path":"models/cerebras-keyformer-lib/modeling_gpt2.py","file_url":"https://github.com/d-matrix-ai/keyformer-llm/blob/HEAD/models/cerebras-keyformer-lib/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bc9fb74e483fb901","mcp_get_code":{"code_sha256":"bc9fb74e483fb901"}},{"arxiv_id":"2403.08293","paper":"/paper/generative-pretrained-structured-transformers","title":"Generative Pretrained Structured Transformers: Unsupervised Syntactic Language Models at Scale","date":"2024-03-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alipay/StructuredLM_RTDT","path":"model/gpt2_flash_attn.py","file_url":"https://github.com/alipay/StructuredLM_RTDT/blob/HEAD/model/gpt2_flash_attn.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3665fea6427d655d","mcp_get_code":{"code_sha256":"3665fea6427d655d"}},{"arxiv_id":"2403.07309","paper":"/paper/reinforced-sequential-decision-making-for","title":"Reinforced Sequential Decision-Making for Sepsis Treatment: The POSNEGDM Framework with Mortality Classifier and Transformer","date":"2024-03-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dipeshtamboli/posnegdm-reinforced-sequential-decision-making-for-sepsis-treatment","path":"dualsight/models/trajectory_gpt2.py","file_url":"https://github.com/dipeshtamboli/posnegdm-reinforced-sequential-decision-making-for-sepsis-treatment/blob/HEAD/dualsight/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2402.10787","paper":"/paper/edgeqat-entropy-and-distribution-guided","title":"EdgeQAT: Entropy and Distribution Guided Quantization-Aware Training for the Acceleration of Lightweight LLMs on the Edge","date":"2024-02-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shawnricecake/edgeqat","path":"distill_train/models/modeling_gpt2_fp16.py","file_url":"https://github.com/shawnricecake/edgeqat/blob/HEAD/distill_train/models/modeling_gpt2_fp16.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3665fea6427d655d","mcp_get_code":{"code_sha256":"3665fea6427d655d"}},{"arxiv_id":"2402.04852","paper":"/paper/multi-patch-prediction-adapting-llms-for-time","title":"Multi-Patch Prediction: Adapting LLMs for Time Series Representation Learning","date":"2024-02-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yxbian23/aLLM4TS","path":"models/modeling_gpt2.py","file_url":"https://github.com/yxbian23/aLLM4TS/blob/HEAD/models/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3665fea6427d655d","mcp_get_code":{"code_sha256":"3665fea6427d655d"}},{"arxiv_id":"2312.16682","paper":"/paper/some-things-are-more-cringe-than-others","title":"Some things are more CRINGE than others: Iterative Preference Optimization with the Pairwise Cringe Loss","date":"2023-12-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/RAM","path":"projects/cocomix/models/modeling_gpt2_cocomix.py","file_url":"https://github.com/facebookresearch/RAM/blob/HEAD/projects/cocomix/models/modeling_gpt2_cocomix.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9a0a23caa628b0b9","mcp_get_code":{"code_sha256":"9a0a23caa628b0b9"}},{"arxiv_id":"2312.13716","paper":"/paper/critic-guided-decision-transformer-for","title":"Critic-Guided Decision Transformer for Offline Reinforcement Learning","date":"2023-12-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sharkwyf/cgdt","path":"decision_transformer/models/trajectory_gpt2.py","file_url":"https://github.com/sharkwyf/cgdt/blob/HEAD/decision_transformer/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2310.19308","paper":"/paper/free-from-bellman-completeness-trajectory","title":"Free from Bellman Completeness: Trajectory Stitching via Model-based Return-conditioned Supervised Learning","date":"2023-10-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhaoyizhou1123/mbrcsl","path":"offlinerlkit/modules/trajectory_gpt2.py","file_url":"https://github.com/zhaoyizhou1123/mbrcsl/blob/HEAD/offlinerlkit/modules/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bc9fb74e483fb901","mcp_get_code":{"code_sha256":"bc9fb74e483fb901"}},{"arxiv_id":"2310.09753","paper":"/paper/when-can-transformers-reason-with-abstract","title":"When can transformers reason with abstract symbols?","date":"2023-10-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eboix/relational-reasoning","path":"train_gpt2/gpt2_with_identity.py","file_url":"https://github.com/eboix/relational-reasoning/blob/HEAD/train_gpt2/gpt2_with_identity.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3665fea6427d655d","mcp_get_code":{"code_sha256":"3665fea6427d655d"}},{"arxiv_id":"2310.09573","paper":"/paper/self-detoxifying-language-models-via","title":"Self-Detoxifying Language Models via Toxification Reversal","date":"2023-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cooperleong00/toxificationreversal","path":"models/gpt2/modeling_gpt2.py","file_url":"https://github.com/cooperleong00/toxificationreversal/blob/HEAD/models/gpt2/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bc9fb74e483fb901","mcp_get_code":{"code_sha256":"bc9fb74e483fb901"}},{"arxiv_id":"2310.09573","paper":"/paper/self-detoxifying-language-models-via","title":"Self-Detoxifying Language Models via Toxification Reversal","date":"2023-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cooperleong00/toxificationreversal","path":"models/gpt2/modeling_gpt2_innerdetox.py","file_url":"https://github.com/cooperleong00/toxificationreversal/blob/HEAD/models/gpt2/modeling_gpt2_innerdetox.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c87b238769f13efb","mcp_get_code":{"code_sha256":"c87b238769f13efb"}},{"arxiv_id":"2308.05061","paper":"/paper/prompting-in-context-operator-learning-with","title":"Fine-Tune Language Models as Multi-Modal Differential Equation Solvers","date":"2023-08-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liuyangmage/in-context-operator-networks","path":"icon-lm/models_gpt2_source.py","file_url":"https://github.com/liuyangmage/in-context-operator-networks/blob/HEAD/icon-lm/models_gpt2_source.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c4bac98236c1fc2c","mcp_get_code":{"code_sha256":"c4bac98236c1fc2c"}},{"arxiv_id":"2305.18459","paper":"/paper/diffusion-model-is-an-effective-planner-and-1","title":"Diffusion Model is an Effective Planner and Data Synthesizer for Multi-Task Reinforcement Learning","date":"2023-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tinnerhrhe/MTDiff","path":"diffuser/models/GPT2.py","file_url":"https://github.com/tinnerhrhe/MTDiff/blob/HEAD/diffuser/models/GPT2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2305.16683","paper":"/paper/future-conditioned-unsupervised-pretraining","title":"Future-conditioned Unsupervised Pretraining for Decision Transformer","date":"2023-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fffffarmer/pdt","path":"src/models/trajectory_gpt2.py","file_url":"https://github.com/fffffarmer/pdt/blob/HEAD/src/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2305.14550","paper":"/paper/2305-14550","title":"When should we prefer Decision Transformers for Offline Reinforcement Learning?","date":"2023-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"prajjwal1/rl_paradigm","path":"exorl/models/trajectory_gpt2.py","file_url":"https://github.com/prajjwal1/rl_paradigm/blob/HEAD/exorl/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2304.07993","paper":"/paper/in-context-operator-learning-for-differential","title":"In-Context Operator Learning with Data Prompts for Differential Equation Problems","date":"2023-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LiuYangMage/in-context-operator-networks","path":"icon-lm/models_gpt2_source.py","file_url":"https://github.com/LiuYangMage/in-context-operator-networks/blob/HEAD/icon-lm/models_gpt2_source.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c4bac98236c1fc2c","mcp_get_code":{"code_sha256":"c4bac98236c1fc2c"}},{"arxiv_id":"2303.07551","paper":"/paper/merging-decision-transformers-weight","title":"Merging Decision Transformers: Weight Averaging for Forming Multi-Task Policies","date":"2023-03-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"daniellawson9999/merging-decision-transformers","path":"decision-transformer/decision_transformer/models/trajectory_gpt2.py","file_url":"https://github.com/daniellawson9999/merging-decision-transformers/blob/HEAD/decision-transformer/decision_transformer/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cb4e99dd4b50f69d","mcp_get_code":{"code_sha256":"cb4e99dd4b50f69d"}},{"arxiv_id":"2212.04501","paper":"/paper/learning-video-representations-from-large","title":"Learning Video Representations from Large Language Models","date":"2022-12-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/lavila","path":"lavila/models/gpt2_gated.py","file_url":"https://github.com/facebookresearch/lavila/blob/HEAD/lavila/models/gpt2_gated.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"bc9fb74e483fb901","mcp_get_code":{"code_sha256":"bc9fb74e483fb901"}},{"arxiv_id":"2212.03506","paper":"/paper/wider-closer-mixture-of-short-channel","title":"WIDER & CLOSER: Mixture of Short-channel Distillers for Zero-shot Cross-lingual Named Entity Recognition","date":"2022-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mckysse/msd","path":"transformers/modeling_gpt2.py","file_url":"https://github.com/mckysse/msd/blob/HEAD/transformers/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"654e91efc679fe84","mcp_get_code":{"code_sha256":"654e91efc679fe84"}},{"arxiv_id":"2211.14655","paper":"/paper/how-crucial-is-transformer-in-decision","title":"How Crucial is Transformer in Decision Transformer?","date":"2022-11-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"max7born/decision-lstm","path":"src/decision_transformer/models/trajectory_gpt2.py","file_url":"https://github.com/max7born/decision-lstm/blob/HEAD/src/decision_transformer/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2211.11694","paper":"/paper/exploring-discrete-diffusion-models-for-image","title":"Exploring Discrete Diffusion Models for Image Captioning","date":"2022-11-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"buxiangzhiren/ddcap","path":"tf_adpt.py","file_url":"https://github.com/buxiangzhiren/ddcap/blob/HEAD/tf_adpt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cb4e99dd4b50f69d","mcp_get_code":{"code_sha256":"cb4e99dd4b50f69d"}},{"arxiv_id":"2211.09817","paper":"/paper/on-the-effect-of-pre-training-for-transformer","title":"On the Effect of Pre-training for Transformer in Different Modality on Offline Reinforcement Learning","date":"2022-11-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"machelreid/can-wikipedia-help-offline-rl","path":"code/decision_transformer/models/trajectory_gpt2.py","file_url":"https://github.com/machelreid/can-wikipedia-help-offline-rl/blob/HEAD/code/decision_transformer/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cb4e99dd4b50f69d","mcp_get_code":{"code_sha256":"cb4e99dd4b50f69d"}},{"arxiv_id":"2210.04839","paper":"/paper/benchmarking-reinforcement-learning-1","title":"Benchmarking Reinforcement Learning Techniques for Autonomous Navigation","date":"2022-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Daffan/ros_jackal","path":"rl_algos/trajectory_gpt2.py","file_url":"https://github.com/Daffan/ros_jackal/blob/HEAD/rl_algos/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2206.11871","paper":"/paper/offline-rl-for-natural-language-generation","title":"Offline RL for Natural Language Generation with Implicit Language Q Learning","date":"2022-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sea-snell/implicit-language-q-learning","path":"src/models/gpt2_optional_final_ln.py","file_url":"https://github.com/sea-snell/implicit-language-q-learning/blob/HEAD/src/models/gpt2_optional_final_ln.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cb4e99dd4b50f69d","mcp_get_code":{"code_sha256":"cb4e99dd4b50f69d"}},{"arxiv_id":"2206.08353","paper":"/paper/towards-understanding-how-machines-can-learn","title":"Towards Understanding How Machines Can Learn Causal Overhypotheses","date":"2022-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cannylab/casual_overhypotheses","path":"models/decision-transformer/models/trajectory_gpt2.py","file_url":"https://github.com/cannylab/casual_overhypotheses/blob/HEAD/models/decision-transformer/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2205.05128","paper":"/paper/human-language-modeling-1","title":"Human Language Modeling","date":"2022-05-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"humanlab/hart","path":"src/model/modeling_hart.py","file_url":"https://github.com/humanlab/hart/blob/HEAD/src/model/modeling_hart.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5235fb1d3272245b","mcp_get_code":{"code_sha256":"5235fb1d3272245b"}},{"arxiv_id":"2202.11705","paper":"/paper/cold-decoding-energy-based-constrained-text","title":"COLD Decoding: Energy-based Constrained Text Generation with Langevin Dynamics","date":"2022-02-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qkaren/COLD_decoding","path":"GPT2ForwardBackward/modeling_opengpt2.py","file_url":"https://github.com/qkaren/COLD_decoding/blob/HEAD/GPT2ForwardBackward/modeling_opengpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2201.06885","paper":"/paper/mining-fine-grained-semantics-via-graph","title":"Evidence-aware Fake News Detection with Graph Neural Networks","date":"2022-01-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CRIPAC-DIG/GET","path":"pytorch_transformers/modeling_gpt2.py","file_url":"https://github.com/CRIPAC-DIG/GET/blob/HEAD/pytorch_transformers/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"654e91efc679fe84","mcp_get_code":{"code_sha256":"654e91efc679fe84"}},{"arxiv_id":"2110.06206","paper":"/paper/starformer-transformer-with-state-action-1","title":"StARformer: Transformer with State-Action-Reward Representations for Visual Reinforcement Learning","date":"2021-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"elicassion/StARformer","path":"gym/models/trajectory_gpt2.py","file_url":"https://github.com/elicassion/StARformer/blob/HEAD/gym/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2106.01345","paper":"/paper/decision-transformer-reinforcement-learning","title":"Decision Transformer: Reinforcement Learning via Sequence Modeling","date":"2021-06-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Amadeus979/decision-transformer","path":"gym/decision_transformer/models/trajectory_gpt2.py","file_url":"https://github.com/Amadeus979/decision-transformer/blob/HEAD/gym/decision_transformer/models/trajectory_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2104.04039","paper":"/paper/plug-and-blend-a-framework-for-controllable","title":"Plug-and-Blend: A Framework for Controllable Story Generation with Blended Control Codes","date":"2021-03-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xxbidiao/plug-and-blend","path":"gedi_helpers/modeling_gpt2.py","file_url":"https://github.com/xxbidiao/plug-and-blend/blob/HEAD/gedi_helpers/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c4bac98236c1fc2c","mcp_get_code":{"code_sha256":"c4bac98236c1fc2c"}},{"arxiv_id":"2010.05607","paper":"/paper/the-elephant-in-the-interpretability-room-why","title":"The elephant in the interpretability room: Why use attention as explanation when we have saliency methods?","date":"2020-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jessevig/bertviz","path":"bertviz/transformers_neuron_view/modeling_gpt2.py","file_url":"https://github.com/jessevig/bertviz/blob/HEAD/bertviz/transformers_neuron_view/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"654e91efc679fe84","mcp_get_code":{"code_sha256":"654e91efc679fe84"}},{"arxiv_id":"2009.12719","paper":"/paper/stylized-dialogue-response-generation-using","title":"Stylized Dialogue Response Generation Using Stylized Unpaired Texts","date":"2020-09-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"silverriver/Stylized_Dialog","path":"TCFC/bt_beam/model/gpt2.py","file_url":"https://github.com/silverriver/Stylized_Dialog/blob/HEAD/TCFC/bt_beam/model/gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c4bac98236c1fc2c","mcp_get_code":{"code_sha256":"c4bac98236c1fc2c"}},{"arxiv_id":"2009.06367","paper":"/paper/gedi-generative-discriminator-guided-sequence","title":"GeDi: Generative Discriminator Guided Sequence Generation","date":"2020-09-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforce/GeDi","path":"modeling_gpt2.py","file_url":"https://github.com/salesforce/GeDi/blob/HEAD/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"c4bac98236c1fc2c","mcp_get_code":{"code_sha256":"c4bac98236c1fc2c"}},{"arxiv_id":"2005.02439","paper":"/paper/contextualizing-hate-speech-classifiers-with","title":"Contextualizing Hate Speech Classifiers with Post-hoc Explanation","date":"2020-05-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"owaisCS/TestHateSpeech","path":"bert/modeling_gpt2.py","file_url":"https://github.com/owaisCS/TestHateSpeech/blob/HEAD/bert/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a62604f618a8cc82","mcp_get_code":{"code_sha256":"a62604f618a8cc82"}},{"arxiv_id":"2005.00796","paper":"/paper/a-simple-language-model-for-task-oriented","title":"A Simple Language Model for Task-Oriented Dialogue","date":"2020-05-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforce/simpletod","path":"models/modeling_gpt2.py","file_url":"https://github.com/salesforce/simpletod/blob/HEAD/models/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"c4bac98236c1fc2c","mcp_get_code":{"code_sha256":"c4bac98236c1fc2c"}},{"arxiv_id":"2005.00558","paper":"/paper/pointer-constrained-text-generation-via","title":"POINTER: Constrained Progressive Text Generation via Insertion-based Generative Pre-training","date":"2020-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dreasysnail/POINTER","path":"pytorch_transformers/modeling_gpt2.py","file_url":"https://github.com/dreasysnail/POINTER/blob/HEAD/pytorch_transformers/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"07bb0220eb5b609f","mcp_get_code":{"code_sha256":"07bb0220eb5b609f"}},{"arxiv_id":"2004.05707","paper":"/paper/vgcn-bert-augmenting-bert-with-graph","title":"VGCN-BERT: Augmenting BERT with Graph Embedding for Text Classification","date":"2020-04-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Louis-udm/VGCN-BERT","path":"old_version/pytorch_pretrained_bert/modeling_gpt2.py","file_url":"https://github.com/Louis-udm/VGCN-BERT/blob/HEAD/old_version/pytorch_pretrained_bert/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8a9df506fe60a0cd","mcp_get_code":{"code_sha256":"8a9df506fe60a0cd"}},{"arxiv_id":"2004.03829","paper":"/paper/exploring-versatile-generative-language-model","title":"Exploring Versatile Generative Language Model Via Parameter-Efficient Transfer Learning","date":"2020-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zlinao/VGLM","path":"pytorch_transformers/modeling_gpt2.py","file_url":"https://github.com/zlinao/VGLM/blob/HEAD/pytorch_transformers/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"654e91efc679fe84","mcp_get_code":{"code_sha256":"654e91efc679fe84"}},{"arxiv_id":"2002.01808","paper":"/paper/k-adapter-infusing-knowledge-into-pre-trained","title":"K-Adapter: Infusing Knowledge into Pre-Trained Models with Adapters","date":"2020-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/K-Adapter","path":"pytorch_transformers/modeling_gpt2.py","file_url":"https://github.com/microsoft/K-Adapter/blob/HEAD/pytorch_transformers/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"654e91efc679fe84","mcp_get_code":{"code_sha256":"654e91efc679fe84"}},{"arxiv_id":"1911.00720","paper":"/paper/zen-pre-training-chinese-text-encoder","title":"ZEN: Pre-training Chinese Text Encoder Enhanced by N-gram Representations","date":"2019-11-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SVAIGBA/TwASP","path":"pytorch_pretrained_bert/modeling_gpt2.py","file_url":"https://github.com/SVAIGBA/TwASP/blob/HEAD/pytorch_pretrained_bert/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a62604f618a8cc82","mcp_get_code":{"code_sha256":"a62604f618a8cc82"}},{"arxiv_id":"1909.05311","paper":"/paper/graph-based-reasoning-over-heterogeneous","title":"Graph-Based Reasoning over Heterogeneous External Knowledge for Commonsense Question Answering","date":"2019-09-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DecstionBack/AAAI_2020_CommonsenseQA","path":"pytorch_transformers/modeling_gpt2.py","file_url":"https://github.com/DecstionBack/AAAI_2020_CommonsenseQA/blob/HEAD/pytorch_transformers/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"07bb0220eb5b609f","mcp_get_code":{"code_sha256":"07bb0220eb5b609f"}},{"arxiv_id":"1908.09355","paper":"/paper/patient-knowledge-distillation-for-bert-model","title":"Patient Knowledge Distillation for BERT Model Compression","date":"2019-08-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Daniel-H-99/Patient-Knowledge-Distillation","path":"BERT/pytorch_pretrained_bert/modeling_gpt2.py","file_url":"https://github.com/Daniel-H-99/Patient-Knowledge-Distillation/blob/HEAD/BERT/pytorch_pretrained_bert/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"a62604f618a8cc82","mcp_get_code":{"code_sha256":"a62604f618a8cc82"}},{"arxiv_id":"1906.08237","paper":"/paper/xlnet-generalized-autoregressive-pretraining","title":"XLNet: Generalized Autoregressive Pretraining for Language Understanding","date":"2019-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"samwisegamjeee/pytorch-transformers","path":"pytorch_transformers/modeling_gpt2.py","file_url":"https://github.com/samwisegamjeee/pytorch-transformers/blob/HEAD/pytorch_transformers/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"07bb0220eb5b609f","mcp_get_code":{"code_sha256":"07bb0220eb5b609f"}},{"arxiv_id":"1906.01698","paper":"/paper/open-sesame-getting-inside-berts-linguistic","title":"Open Sesame: Getting Inside BERT's Linguistic Knowledge","date":"2019-06-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yongjie-lin/bert-opensesame","path":"bertviz/bertviz/pytorch_pretrained_bert/modeling_gpt2.py","file_url":"https://github.com/yongjie-lin/bert-opensesame/blob/HEAD/bertviz/bertviz/pytorch_pretrained_bert/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a62604f618a8cc82","mcp_get_code":{"code_sha256":"a62604f618a8cc82"}},{"arxiv_id":"aaai_33874","paper":null,"title":"arXiv:aaai_33874","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"txsun1997/Black-Box-Tuning","path":"models/deep_modeling_gpt2.py","file_url":"https://github.com/txsun1997/Black-Box-Tuning/blob/HEAD/models/deep_modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00a33466c69c5705","mcp_get_code":{"code_sha256":"00a33466c69c5705"}},{"arxiv_id":"2025.emnlp-main.688","paper":null,"title":"arXiv:2025.emnlp-main.688","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"yueen-ma/Astra","path":"model/trajectory_transformer.py","file_url":"https://github.com/yueen-ma/Astra/blob/HEAD/model/trajectory_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3665fea6427d655d","mcp_get_code":{"code_sha256":"3665fea6427d655d"}},{"arxiv_id":"2024.findings-naacl.32","paper":null,"title":"arXiv:2024.findings-naacl.32","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"SnowYJ/sem_syn_separation","path":"optimus_separate_graph_sem_syntax_fuse_gpt2/pytorch_transformers/modeling_gpt2.py","file_url":"https://github.com/SnowYJ/sem_syn_separation/blob/HEAD/optimus_separate_graph_sem_syntax_fuse_gpt2/pytorch_transformers/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"654e91efc679fe84","mcp_get_code":{"code_sha256":"654e91efc679fe84"}},{"arxiv_id":"2023.findings-acl.765","paper":null,"title":"arXiv:2023.findings-acl.765","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"qtli/EIB","path":"code/EIB_model/modeling_gpt2.py","file_url":"https://github.com/qtli/EIB/blob/HEAD/code/EIB_model/modeling_gpt2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"59e7bfe60606f53b","mcp_get_code":{"code_sha256":"59e7bfe60606f53b"}}]}