{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/dot-product-attention","entry":"dot_product_attention","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":15,"n_papers_ran":7,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":10,"n_samples_ran":5,"n_samples_fingerprinted":1,"n_places":15,"n_places_pointer_only":3,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":2,"ran":3,"unverified":5},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2501.06999","paper":"/paper/likelihood-training-of-cascaded-diffusion","title":"Likelihood Training of Cascaded Diffusion Models via Hierarchical Volume-preserving Maps","date":"2025-01-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lihenryhfl/pcdm","path":"nn.py","file_url":"https://github.com/lihenryhfl/pcdm/blob/HEAD/nn.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3650a2dfe2e97298","mcp_get_code":{"code_sha256":"3650a2dfe2e97298"}},{"arxiv_id":"2410.10012","paper":"/paper/naraim-native-aspect-ratio-autoregressive","title":"NARAIM: Native Aspect Ratio Autoregressive Image Models","date":"2024-10-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"daniel-gallo/naraim","path":"model/pretraining.py","file_url":"https://github.com/daniel-gallo/naraim/blob/HEAD/model/pretraining.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d1eecbe4982a95d1","mcp_get_code":{"code_sha256":"d1eecbe4982a95d1"}},{"arxiv_id":"2406.13476","paper":"/paper/llms-are-zero-shot-context-aware-simultaneous","title":"LLMs Are Zero-Shot Context-Aware Simultaneous Translators","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RomanKoshkin/toLLMatch","path":"whisper-jax/whisper_jax/layers.py","file_url":"https://github.com/RomanKoshkin/toLLMatch/blob/HEAD/whisper-jax/whisper_jax/layers.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2000f34f95c683d6","mcp_get_code":{"code_sha256":"2000f34f95c683d6"}},{"arxiv_id":"2403.00043","paper":"/paper/rinalmo-general-purpose-rna-language-models","title":"RiNALMo: General-Purpose RNA Language Models Can Generalize Well on Structure Prediction Tasks","date":"2024-02-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lbcb-sci/rinalmo","path":"rinalmo/model/attention.py","file_url":"https://github.com/lbcb-sci/rinalmo/blob/HEAD/rinalmo/model/attention.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f6afd2235cdb7855","mcp_get_code":{"code_sha256":"f6afd2235cdb7855"}},{"arxiv_id":"2308.06912","paper":"/paper/causallm-is-not-optimal-for-in-context","title":"CausalLM is not optimal for in-context learning","date":"2023-08-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"google-research/causallm_icl","path":"self_attention_patch.py","file_url":"https://github.com/google-research/causallm_icl/blob/HEAD/self_attention_patch.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"25ce8c68e697049c","mcp_get_code":{"code_sha256":"25ce8c68e697049c"}},{"arxiv_id":"2305.03935","paper":"/paper/improved-techniques-for-maximum-likelihood","title":"Improved Techniques for Maximum Likelihood Estimation for Diffusion ODEs","date":"2023-05-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-ml/i-dode","path":"model_vdm.py","file_url":"https://github.com/thu-ml/i-dode/blob/HEAD/model_vdm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f25b8c5dd6fba4a7","mcp_get_code":{"code_sha256":"f25b8c5dd6fba4a7"}},{"arxiv_id":"2303.16133","paper":"/paper/exposing-and-addressing-cross-task","title":"Exposing and Addressing Cross-Task Inconsistency in Unified Vision-Language Models","date":"2023-03-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"adymaharana/cococon","path":"unified-io/uio/t5x_layers.py","file_url":"https://github.com/adymaharana/cococon/blob/HEAD/unified-io/uio/t5x_layers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"dac0288079ac279f","mcp_get_code":{"code_sha256":"dac0288079ac279f"}},{"arxiv_id":"2212.04356","paper":"/paper/robust-speech-recognition-via-large-scale-1","title":"Robust Speech Recognition via Large-Scale Weak Supervision","date":"2022-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sanchit-gandhi/whisper-jax","path":"whisper_jax/layers.py","file_url":"https://github.com/sanchit-gandhi/whisper-jax/blob/HEAD/whisper_jax/layers.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"2000f34f95c683d6","mcp_get_code":{"code_sha256":"2000f34f95c683d6"}},{"arxiv_id":"2206.05408","paper":"/paper/multi-instrument-music-synthesis-with","title":"Multi-instrument Music Synthesis with Spectrogram Diffusion","date":"2022-06-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"magenta/music-spectrogram-diffusion","path":"music_spectrogram_diffusion/layers.py","file_url":"https://github.com/magenta/music-spectrogram-diffusion/blob/HEAD/music_spectrogram_diffusion/layers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"dac0288079ac279f","mcp_get_code":{"code_sha256":"dac0288079ac279f"}},{"arxiv_id":"2205.01580","paper":"/paper/better-plain-vit-baselines-for-imagenet-1k","title":"Better plain ViT baselines for ImageNet-1k","date":"2022-05-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuyangshu/retinavit","path":"big_vision/models/attn_override.py","file_url":"https://github.com/yuyangshu/retinavit/blob/HEAD/big_vision/models/attn_override.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2c4ed3f75ab6ee14","mcp_get_code":{"code_sha256":"2c4ed3f75ab6ee14"}},{"arxiv_id":"2112.00791","paper":"/paper/controlling-conditional-language-models-with","title":"Controlling Conditional Language Models without Catastrophic Forgetting","date":"2021-12-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"google/t5patches","path":"t5patches/layers.py","file_url":"https://github.com/google/t5patches/blob/HEAD/t5patches/layers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"dac0288079ac279f","mcp_get_code":{"code_sha256":"dac0288079ac279f"}},{"arxiv_id":"2107.00630","paper":"/paper/variational-diffusion-models","title":"Variational Diffusion Models","date":"2021-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"google-research/vdm","path":"model_vdm.py","file_url":"https://github.com/google-research/vdm/blob/HEAD/model_vdm.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3650a2dfe2e97298","mcp_get_code":{"code_sha256":"3650a2dfe2e97298"}},{"arxiv_id":"2106.03598","paper":"/paper/scifive-a-text-to-text-transformer-model-for","title":"SciFive: a text-to-text transformer model for biomedical literature","date":"2021-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"justinphan3110/SciFive","path":"biot5x/configs/t5/layers.py","file_url":"https://github.com/justinphan3110/SciFive/blob/HEAD/biot5x/configs/t5/layers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dac0288079ac279f","mcp_get_code":{"code_sha256":"dac0288079ac279f"}},{"arxiv_id":"2005.04732","paper":"/paper/towards-robustifying-nli-models-against","title":"Towards Robustifying NLI Models Against Lexical Dataset Biases","date":"2020-05-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"owenzx/LexicalDebias-ACL2020","path":"models/hex.py","file_url":"https://github.com/owenzx/LexicalDebias-ACL2020/blob/HEAD/models/hex.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0403fe04af127ad4","mcp_get_code":{"code_sha256":"0403fe04af127ad4"}},{"arxiv_id":"1909.11942","paper":"/paper/albert-a-lite-bert-for-self-supervised","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","date":"2019-09-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"brightmart/albert_zh","path":"modeling_google.py","file_url":"https://github.com/brightmart/albert_zh/blob/HEAD/modeling_google.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"847351b08a6a93c6","mcp_get_code":{"code_sha256":"847351b08a6a93c6"}}]}