{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/scaled-dot-product-attention","entry":"scaled_dot_product_attention","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":43,"n_papers_ran":24,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":52,"n_samples_ran":32,"n_samples_fingerprinted":13,"n_places":55,"n_places_pointer_only":21,"by_status":{"ran_honours":4,"ran_violates":4,"ran_draft_wrong":8,"ran_fixture":6,"ran":10,"unverified":20},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.30116","paper":"/paper/arxiv-2605-30116","title":"SGMD: Score Gradient Matching Distillation for Few-Step Video Diffusion Distillation","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"ModelTC/LightX2V","path":"lightx2v/utils/print_atten_score.py","file_url":"https://github.com/ModelTC/LightX2V/blob/HEAD/lightx2v/utils/print_atten_score.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"68fcdfd881d93961","mcp_get_code":{"code_sha256":"68fcdfd881d93961"}},{"arxiv_id":"2605.30022","paper":"/paper/arxiv-2605-30022","title":"Give it Space! Explicit Disentangling of Positional and Semantic Representations in Encoders","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"LequeuISIR/DSTG-encoder","path":"src/neobert/model/model.py","file_url":"https://github.com/LequeuISIR/DSTG-encoder/blob/HEAD/src/neobert/model/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"13c8fb0873f97a00","mcp_get_code":{"code_sha256":"13c8fb0873f97a00"}},{"arxiv_id":"2604.27037","paper":"/paper/arxiv-2604-27037","title":"Hypencoder Revisited: Reproducibility and Analysis of Non-Linear Scoring for First-Stage Retrieval","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"arneeichholtz/Hypencoder-reprod","path":"hypencoder_cb/modeling/hypencoder.py","file_url":"https://github.com/arneeichholtz/Hypencoder-reprod/blob/HEAD/hypencoder_cb/modeling/hypencoder.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6e979b51a5011c36","mcp_get_code":{"code_sha256":"6e979b51a5011c36"}},{"arxiv_id":"2604.15750","paper":"/paper/arxiv-2604-15750","title":"DepCap: Adaptive Block-Wise Parallel Decoding for Efficient Diffusion LM Inference","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"X-Xia0828/DepCap","path":"llada/model/modeling_llada.py","file_url":"https://github.com/X-Xia0828/DepCap/blob/HEAD/llada/model/modeling_llada.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a1e20a66389a45ff","mcp_get_code":{"code_sha256":"a1e20a66389a45ff"}},{"arxiv_id":"2603.14366","paper":"/paper/arxiv-2603-14366","title":"Representation Alignment for Just Image Transformers is not Easier than You Think","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"kaist-cvml/PixelREPA","path":"model_pixelREPA.py","file_url":"https://github.com/kaist-cvml/PixelREPA/blob/HEAD/model_pixelREPA.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ddbfee8e56c96b9c","mcp_get_code":{"code_sha256":"ddbfee8e56c96b9c"}},{"arxiv_id":"2601.21484","paper":"/paper/arxiv-2601-21484","title":"ETS: Energy-Guided Test-Time Scaling for Training-Free RL Alignment","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"sheriyuo/ETS","path":"llada/model/modeling_llada.py","file_url":"https://github.com/sheriyuo/ETS/blob/HEAD/llada/model/modeling_llada.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a1e20a66389a45ff","mcp_get_code":{"code_sha256":"a1e20a66389a45ff"}},{"arxiv_id":"2510.23111","paper":"/paper/arxiv-2510-23111","title":"Neural Emulator Superiority: When Machine Learning for PDEs Surpasses its Training Data","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"tum-pbs/emulator-superiority","path":"code/experiments/transformer.py","file_url":"https://github.com/tum-pbs/emulator-superiority/blob/HEAD/code/experiments/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f0a2fdc3bd45affc","mcp_get_code":{"code_sha256":"f0a2fdc3bd45affc"}},{"arxiv_id":"2510.15783","paper":"/paper/arxiv-2510-15783","title":"ReCon: Region-Controllable Data Augmentation with Rectification and Alignment for Object Detection","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"haoweiz23/ReCon","path":"models/attention_processor.py","file_url":"https://github.com/haoweiz23/ReCon/blob/HEAD/models/attention_processor.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f759aa4f93d14c47","mcp_get_code":{"code_sha256":"f759aa4f93d14c47"}},{"arxiv_id":"2506.21416","paper":"/paper/xverse-consistent-multi-subject-control-of","title":"XVerse: Consistent Multi-Subject Control of Identity and Semantic Attributes via DiT Modulation","date":"2025-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/xverse","path":"src/flux/block.py","file_url":"https://github.com/bytedance/xverse/blob/HEAD/src/flux/block.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"75e0eb0c22ca129e","mcp_get_code":{"code_sha256":"75e0eb0c22ca129e"}},{"arxiv_id":"2505.24244","paper":"/paper/mamba-knockout-for-unraveling-factual","title":"Mamba Knockout for Unraveling Factual Information Flow","date":"2025-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nirendy/mamba-knockout","path":"src/experiments/knockout/llama/scaled_dot_product_attention.py","file_url":"https://github.com/nirendy/mamba-knockout/blob/HEAD/src/experiments/knockout/llama/scaled_dot_product_attention.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bec2160c46dd3280","mcp_get_code":{"code_sha256":"bec2160c46dd3280"}},{"arxiv_id":"2503.03751","paper":"/paper/gen3c-3d-informed-world-consistent-video","title":"GEN3C: 3D-Informed World-Consistent Video Generation with Precise Camera Control","date":"2025-03-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nv-tlabs/GEN3C","path":"cosmos_predict1/autoregressive/modules/attention.py","file_url":"https://github.com/nv-tlabs/GEN3C/blob/HEAD/cosmos_predict1/autoregressive/modules/attention.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"abc31b2f673701c7","mcp_get_code":{"code_sha256":"abc31b2f673701c7"}},{"arxiv_id":"2502.17437","paper":"/paper/fractal-generative-models","title":"Fractal Generative Models","date":"2025-02-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LTH14/fractalgen","path":"models/ar.py","file_url":"https://github.com/LTH14/fractalgen/blob/HEAD/models/ar.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d2525f74659345d8","mcp_get_code":{"code_sha256":"d2525f74659345d8"}},{"arxiv_id":"2502.04320","paper":"/paper/conceptattention-diffusion-transformers-learn","title":"ConceptAttention: Diffusion Transformers Learn Highly Interpretable Features","date":"2025-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"helblazer811/ConceptAttention","path":"concept_attention/flux/dit_block.py","file_url":"https://github.com/helblazer811/ConceptAttention/blob/HEAD/concept_attention/flux/dit_block.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1af01c699046b8d4","mcp_get_code":{"code_sha256":"1af01c699046b8d4"}},{"arxiv_id":"2501.03575","paper":"/paper/cosmos-world-foundation-model-platform-for","title":"Cosmos World Foundation Model Platform for Physical AI","date":"2025-01-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nvidia-cosmos/cosmos-predict1","path":"cosmos_predict1/autoregressive/modules/attention.py","file_url":"https://github.com/nvidia-cosmos/cosmos-predict1/blob/HEAD/cosmos_predict1/autoregressive/modules/attention.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"abc31b2f673701c7","mcp_get_code":{"code_sha256":"abc31b2f673701c7"}},{"arxiv_id":"2501.01986","paper":"/paper/framefusion-combining-similarity-and","title":"FrameFusion: Combining Similarity and Importance for Video Token Reduction on Large Visual Language Models","date":"2024-12-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-nics/framefusion","path":"framefusion/utils.py","file_url":"https://github.com/thu-nics/framefusion/blob/HEAD/framefusion/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2e002934a51f4e51","mcp_get_code":{"code_sha256":"2e002934a51f4e51"}},{"arxiv_id":"2410.13846","paper":"/paper/simlayerkv-a-simple-framework-for-layer-level","title":"SimLayerKV: A Simple Framework for Layer-Level KV Cache Reduction","date":"2024-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sail-sg/simlayerkv","path":"LongBench/SimLayerKV_attention_llama.py","file_url":"https://github.com/sail-sg/simlayerkv/blob/HEAD/LongBench/SimLayerKV_attention_llama.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"15435bf612fd7f6f","mcp_get_code":{"code_sha256":"15435bf612fd7f6f"}},{"arxiv_id":"2410.13846","paper":"/paper/simlayerkv-a-simple-framework-for-layer-level","title":"SimLayerKV: A Simple Framework for Layer-Level KV Cache Reduction","date":"2024-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sail-sg/simlayerkv","path":"LongBench/SimLayerKV_attention_qwen.py","file_url":"https://github.com/sail-sg/simlayerkv/blob/HEAD/LongBench/SimLayerKV_attention_qwen.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"996f634b37f61510","mcp_get_code":{"code_sha256":"996f634b37f61510"}},{"arxiv_id":"2409.12319","paper":"/paper/large-language-models-are-strong-audio-visual","title":"Large Language Models are Strong Audio-Visual Speech Recognition Learners","date":"2024-09-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"umbertocappellazzo/llama-avsr","path":"models/Llama_LoRA.py","file_url":"https://github.com/umbertocappellazzo/llama-avsr/blob/HEAD/models/Llama_LoRA.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f52b1ca383922f4e","mcp_get_code":{"code_sha256":"f52b1ca383922f4e"}},{"arxiv_id":"2409.11899","paper":"/paper/multi-grid-graph-neural-networks-with-self","title":"Multi-Grid Graph Neural Networks with Self-Attention for Computational Mechanics","date":"2024-09-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DonsetPG/graph-physics","path":"jraphphysics/models/layers.py","file_url":"https://github.com/DonsetPG/graph-physics/blob/HEAD/jraphphysics/models/layers.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"417486e192f1d2e2","mcp_get_code":{"code_sha256":"417486e192f1d2e2"}},{"arxiv_id":"2406.12454","paper":"/paper/a-neural-column-generation-approach-to-the","title":"A Neural Column Generation Approach to the Vehicle Routing Problem with Two-Dimensional Loading and Last-In-First-Out Constraints","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xyfffff/NCG-for-2L-CVRP","path":"bpp/model.py","file_url":"https://github.com/xyfffff/NCG-for-2L-CVRP/blob/HEAD/bpp/model.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e799361645d45d27","mcp_get_code":{"code_sha256":"e799361645d45d27"}},{"arxiv_id":"2406.02550","paper":"/paper/learning-to-grok-emergence-of-in-context","title":"Learning to grok: Emergence of in-context learning and skill composition in modular arithmetic tasks","date":"2024-06-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ablghtianyi/ICL_Modular_Arithmetic","path":"interpretability/attn_map.py","file_url":"https://github.com/ablghtianyi/ICL_Modular_Arithmetic/blob/HEAD/interpretability/attn_map.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4f3bdb6c844ac5cc","mcp_get_code":{"code_sha256":"4f3bdb6c844ac5cc"}},{"arxiv_id":"2405.09789","paper":"/paper/lemevit-efficient-vision-transformer-with","title":"LeMeViT: Efficient Vision Transformer with Learnable Meta Tokens for Remote Sensing Image Interpretation","date":"2024-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ViTAE-Transformer/LeMeViT","path":"models/lemevit.py","file_url":"https://github.com/ViTAE-Transformer/LeMeViT/blob/HEAD/models/lemevit.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1a1001adc58b631f","mcp_get_code":{"code_sha256":"1a1001adc58b631f"}},{"arxiv_id":"2404.19563","paper":"/paper/repeval-effective-text-evaluation-with-llm","title":"RepEval: Effective Text Evaluation with LLM Representation","date":"2024-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shikib/usr","path":"transformers/modeling_ctrl.py","file_url":"https://github.com/shikib/usr/blob/HEAD/transformers/modeling_ctrl.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"57ebc1447d921c46","mcp_get_code":{"code_sha256":"57ebc1447d921c46"}},{"arxiv_id":"2401.12652","paper":"/paper/from-numbers-to-words-multi-modal-bankruptcy","title":"From Numbers to Words: Multi-Modal Bankruptcy Prediction Using the ECL Dataset","date":null,"month_inferred_from_arxiv_id":"2024-01","title_source":"archive","repo":"henriarnoug/ECL","path":"sentence-attention/model_utils.py","file_url":"https://github.com/henriarnoug/ECL/blob/HEAD/sentence-attention/model_utils.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"819a93855d130bbc","mcp_get_code":{"code_sha256":"819a93855d130bbc"}},{"arxiv_id":"2312.04147","paper":"/paper/an-improved-masking-strategy-for-self","title":"An Improved Masking Strategy for Self-supervised Masked Reconstruction in Human Activity Recognition","date":null,"month_inferred_from_arxiv_id":"2023-12","title_source":"archive","repo":"diheal/channle_masking","path":"multiHeadAttention.py","file_url":"https://github.com/diheal/channle_masking/blob/HEAD/multiHeadAttention.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"00c2ece4e478acf2","mcp_get_code":{"code_sha256":"00c2ece4e478acf2"}},{"arxiv_id":"2308.11358","paper":"/paper/how-much-temporal-long-term-context-is-needed","title":"How Much Temporal Long-Term Context is Needed for Action Segmentation?","date":"2023-08-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltcontext/ltcontext","path":"ltc/model/attention_utils.py","file_url":"https://github.com/ltcontext/ltcontext/blob/HEAD/ltc/model/attention_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"286e98f2ad765c4b","mcp_get_code":{"code_sha256":"286e98f2ad765c4b"}},{"arxiv_id":"2212.03506","paper":"/paper/wider-closer-mixture-of-short-channel","title":"WIDER & CLOSER: Mixture of Short-channel Distillers for Zero-shot Cross-lingual Named Entity Recognition","date":"2022-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mckysse/msd","path":"transformers/modeling_ctrl.py","file_url":"https://github.com/mckysse/msd/blob/HEAD/transformers/modeling_ctrl.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"57ebc1447d921c46","mcp_get_code":{"code_sha256":"57ebc1447d921c46"}},{"arxiv_id":"2211.14730","paper":"/paper/a-time-series-is-worth-64-words-long-term","title":"A Time Series is Worth 64 Words: Long-term Forecasting with Transformers","date":"2022-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"romilbert/samformer","path":"samformer_pytorch/samformer/samformer.py","file_url":"https://github.com/romilbert/samformer/blob/HEAD/samformer_pytorch/samformer/samformer.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"21d036b82013b9f5","mcp_get_code":{"code_sha256":"21d036b82013b9f5"}},{"arxiv_id":"2205.01286","paper":"/paper/when-multi-level-meets-multi-interest-a-multi","title":"When Multi-Level Meets Multi-Interest: A Multi-Grained Neural Model for Sequential Recommendation","date":"2022-05-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"whuir/mgnm","path":"code/util_se.py","file_url":"https://github.com/whuir/mgnm/blob/HEAD/code/util_se.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8fe64623b4d2875f","mcp_get_code":{"code_sha256":"8fe64623b4d2875f"}},{"arxiv_id":"2110.10054","paper":"/paper/generating-symbolic-reasoning-problems-with-1","title":"Generating Symbolic Reasoning Problems with Transformer GANs","date":"2021-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"reactive-systems/TGAN-SR","path":"impl/tgan_sr/transformer/attention.py","file_url":"https://github.com/reactive-systems/TGAN-SR/blob/HEAD/impl/tgan_sr/transformer/attention.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6e22d18d71e35c49","mcp_get_code":{"code_sha256":"6e22d18d71e35c49"}},{"arxiv_id":"2107.05916","paper":"/paper/towards-automatic-instrumentation-by-learning","title":"Towards Automatic Instrumentation by Learning to Separate Parts in Symbolic Multitrack Music","date":"2021-07-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salu133445/arranger","path":"arranger/transformer/model.py","file_url":"https://github.com/salu133445/arranger/blob/HEAD/arranger/transformer/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cec70fbe641ff7aa","mcp_get_code":{"code_sha256":"cec70fbe641ff7aa"}},{"arxiv_id":"2106.08185","paper":"/paper/kernel-identification-through-transformers","title":"Kernel Identification Through Transformers","date":"2021-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"frgsimpson/kitt","path":"kitt/networks/transformer/set_transformer_blocks.py","file_url":"https://github.com/frgsimpson/kitt/blob/HEAD/kitt/networks/transformer/set_transformer_blocks.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"66755ac81353d809","mcp_get_code":{"code_sha256":"66755ac81353d809"}},{"arxiv_id":"2106.02097","paper":"/paper/a-consciousness-inspired-planning-agent-for","title":"A Consciousness-Inspired Planning Agent for Model-Based Reinforcement Learning","date":"2021-06-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"PwnerHarry/CP","path":"components_CP.py","file_url":"https://github.com/PwnerHarry/CP/blob/HEAD/components_CP.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cd1c640990b1df7e","mcp_get_code":{"code_sha256":"cd1c640990b1df7e"}},{"arxiv_id":"2104.05704","paper":"/paper/escaping-the-big-data-paradigm-with-compact","title":"Escaping the Big Data Paradigm with Compact Transformers","date":"2021-04-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Ryul0rd/compact-convolutional-transformer","path":"compact_conv_transformer.py","file_url":"https://github.com/Ryul0rd/compact-convolutional-transformer/blob/HEAD/compact_conv_transformer.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f35b452496f8552c","mcp_get_code":{"code_sha256":"f35b452496f8552c"}},{"arxiv_id":"2008.08692","paper":"/paper/enhancing-graph-neural-network-based-fraud","title":"Enhancing Graph Neural Network-based Fraud Detectors against Camouflaged Fraudsters","date":"2020-08-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"safe-graph/dgfraud-tf2","path":"layers/layers.py","file_url":"https://github.com/safe-graph/dgfraud-tf2/blob/HEAD/layers/layers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"00245cbc1ab1f183","mcp_get_code":{"code_sha256":"00245cbc1ab1f183"}},{"arxiv_id":"2003.04218","paper":"/paper/teaching-temporal-logics-to-neural-networks","title":"Teaching Temporal Logics to Neural Networks","date":"2020-03-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"reactive-systems/deepltl","path":"deepltl/layers/attention.py","file_url":"https://github.com/reactive-systems/deepltl/blob/HEAD/deepltl/layers/attention.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6ac8c2e8189b5ab7","mcp_get_code":{"code_sha256":"6ac8c2e8189b5ab7"}},{"arxiv_id":"2003.04218","paper":"/paper/teaching-temporal-logics-to-neural-networks","title":"Teaching Temporal Logics to Neural Networks","date":"2020-03-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"necrashter/deepltl-pytorch","path":"deepltl/layers/attention.py","file_url":"https://github.com/necrashter/deepltl-pytorch/blob/HEAD/deepltl/layers/attention.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"73b90c22738f0032","mcp_get_code":{"code_sha256":"73b90c22738f0032"}},{"arxiv_id":"2002.04745","paper":"/paper/on-layer-normalization-in-the-transformer-1","title":"On Layer Normalization in the Transformer Architecture","date":"2020-02-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"colorfulscoop/tfdlg","path":"tfdlg/models.py","file_url":"https://github.com/colorfulscoop/tfdlg/blob/HEAD/tfdlg/models.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3864ff9606edf68a","mcp_get_code":{"code_sha256":"3864ff9606edf68a"}},{"arxiv_id":"1910.14599","paper":"/paper/adversarial-nli-a-new-benchmark-for-natural","title":"Adversarial NLI: A New Benchmark for Natural Language Understanding","date":"2019-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tlatkowski/multihead-siamese-nets","path":"layers/attention.py","file_url":"https://github.com/tlatkowski/multihead-siamese-nets/blob/HEAD/layers/attention.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f912d2bb7968ce7f","mcp_get_code":{"code_sha256":"f912d2bb7968ce7f"}},{"arxiv_id":"1909.05858","paper":"/paper/ctrl-a-conditional-transformer-language-model-1","title":"CTRL: A Conditional Transformer Language Model for Controllable Generation","date":"2019-09-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"UKPLab/controlled-argument-generation","path":"transformer.py","file_url":"https://github.com/UKPLab/controlled-argument-generation/blob/HEAD/transformer.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"c37ff914e4e7a62e","mcp_get_code":{"code_sha256":"c37ff914e4e7a62e"}},{"arxiv_id":"1909.05858","paper":"/paper/ctrl-a-conditional-transformer-language-model-1","title":"CTRL: A Conditional Transformer Language Model for Controllable Generation","date":"2019-09-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforce/ctrl","path":"pytorch_transformer.py","file_url":"https://github.com/salesforce/ctrl/blob/HEAD/pytorch_transformer.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"6da14d7cf9778239","mcp_get_code":{"code_sha256":"6da14d7cf9778239"}},{"arxiv_id":"1904.10509","paper":"/paper/190410509","title":"Generating Long Sequences with Sparse Transformers","date":"2019-04-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wilson1yan/VideoGPT","path":"videogpt/attention.py","file_url":"https://github.com/wilson1yan/VideoGPT/blob/HEAD/videogpt/attention.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9d35276799c65577","mcp_get_code":{"code_sha256":"9d35276799c65577"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rajlm10/Chandler","path":"utils.py","file_url":"https://github.com/rajlm10/Chandler/blob/HEAD/utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"aa2d3d2c35ae9b63","mcp_get_code":{"code_sha256":"aa2d3d2c35ae9b63"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"anhtu293/transformer_from_scratch","path":"model/transformer.py","file_url":"https://github.com/anhtu293/transformer_from_scratch/blob/HEAD/model/transformer.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c7e7c016c66eb84d","mcp_get_code":{"code_sha256":"c7e7c016c66eb84d"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"antoinecollas/transformer_neural_machine_translation","path":"transformer/transformer.py","file_url":"https://github.com/antoinecollas/transformer_neural_machine_translation/blob/HEAD/transformer/transformer.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6aa80807c7eb13aa","mcp_get_code":{"code_sha256":"6aa80807c7eb13aa"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"soumik12345/transformer.pytorch","path":"src/model.py","file_url":"https://github.com/soumik12345/transformer.pytorch/blob/HEAD/src/model.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5796ea2aeed41947","mcp_get_code":{"code_sha256":"5796ea2aeed41947"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dhiraa/tener","path":"src/tener/models/vanialla_transformer.py","file_url":"https://github.com/dhiraa/tener/blob/HEAD/src/tener/models/vanialla_transformer.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"28a1ce930e92de0c","mcp_get_code":{"code_sha256":"28a1ce930e92de0c"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"abhaskumarsinha/MinimalGPT","path":"GPT.py","file_url":"https://github.com/abhaskumarsinha/MinimalGPT/blob/HEAD/GPT.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5d28f773b90ce3af","mcp_get_code":{"code_sha256":"5d28f773b90ce3af"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"abhaskumarsinha/Keras-implementation-of-Transformer-Architecture","path":"Transformer.py","file_url":"https://github.com/abhaskumarsinha/Keras-implementation-of-Transformer-Architecture/blob/HEAD/Transformer.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"4e705f4378907830","mcp_get_code":{"code_sha256":"4e705f4378907830"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Aveek-Saha/Transformer","path":"model.py","file_url":"https://github.com/Aveek-Saha/Transformer/blob/HEAD/model.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"547a0b322775d240","mcp_get_code":{"code_sha256":"547a0b322775d240"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Maple728/transformer","path":"models/transformer.py","file_url":"https://github.com/Maple728/transformer/blob/HEAD/models/transformer.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7fe06ef157b00d63","mcp_get_code":{"code_sha256":"7fe06ef157b00d63"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Kyubyong/transformer","path":"model.py","file_url":"https://github.com/Kyubyong/transformer/blob/HEAD/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0dcc4910e658accc","mcp_get_code":{"code_sha256":"0dcc4910e658accc"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"majing2019/transformer","path":"model.py","file_url":"https://github.com/majing2019/transformer/blob/HEAD/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"57e1b33a663f6ac6","mcp_get_code":{"code_sha256":"57e1b33a663f6ac6"}},{"arxiv_id":"1704.04368","paper":"/paper/get-to-the-point-summarization-with-pointer","title":"Get To The Point: Summarization with Pointer-Generator Networks","date":"2017-04-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"steph1793/Pointer_Transformer_Generator","path":"transformer.py","file_url":"https://github.com/steph1793/Pointer_Transformer_Generator/blob/HEAD/transformer.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"70c86dc3686d36b5","mcp_get_code":{"code_sha256":"70c86dc3686d36b5"}},{"arxiv_id":"2021.acl-long.390","paper":null,"title":"arXiv:2021.acl-long.390","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"thunlp/MetaAdaptRank","path":"contrastqg/transformers/modeling_ctrl.py","file_url":"https://github.com/thunlp/MetaAdaptRank/blob/HEAD/contrastqg/transformers/modeling_ctrl.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"27e71cce0a74b944","mcp_get_code":{"code_sha256":"27e71cce0a74b944"}}]}