{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/transformer-model","entry":"transformer_model","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":7,"n_papers_ran":0,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":26,"n_samples_ran":0,"n_samples_fingerprinted":0,"n_places":26,"n_places_pointer_only":5,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":0,"unverified":26},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2009.07258","paper":"/paper/bert-qe-contextualized-query-expansion-for","title":"BERT-QE: Contextualized Query Expansion for Document Re-ranking","date":"2020-09-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zh-zheng/BERT-QE","path":"functions.py","file_url":"https://github.com/zh-zheng/BERT-QE/blob/HEAD/functions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5b85e1462c909ec8","mcp_get_code":{"code_sha256":"5b85e1462c909ec8"}},{"arxiv_id":"2004.11579","paper":"/paper/probabilistically-masked-language-model","title":"Probabilistically Masked Language Model Capable of Autoregressive Generation in Arbitrary Word Order","date":"2020-04-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"huawei-noah/Pretrained-Language-Model","path":"PMLM/interactive_conditional_samples_sincos_acrostic.py","file_url":"https://github.com/huawei-noah/Pretrained-Language-Model/blob/HEAD/PMLM/interactive_conditional_samples_sincos_acrostic.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bc3e0393640c7f77","mcp_get_code":{"code_sha256":"bc3e0393640c7f77"}},{"arxiv_id":"1909.11942","paper":"/paper/albert-a-lite-bert-for-self-supervised","title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations","date":"2019-09-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"brightmart/albert_zh","path":"modeling_google.py","file_url":"https://github.com/brightmart/albert_zh/blob/HEAD/modeling_google.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e447e4b29a5d024b","mcp_get_code":{"code_sha256":"e447e4b29a5d024b"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TeamLab/bert-gcn-for-paper-citation","path":"modeling.py","file_url":"https://github.com/TeamLab/bert-gcn-for-paper-citation/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"dbc9628a5ca3b1f5","mcp_get_code":{"code_sha256":"dbc9628a5ca3b1f5"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DeligientSloth/bert-tensorflow","path":"modeling.py","file_url":"https://github.com/DeligientSloth/bert-tensorflow/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e353d671edbdf34f","mcp_get_code":{"code_sha256":"e353d671edbdf34f"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wchh127/yykf","path":"modeling.py","file_url":"https://github.com/wchh127/yykf/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"79608bac19b3ff5c","mcp_get_code":{"code_sha256":"79608bac19b3ff5c"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Satan012/BERT","path":"modeling.py","file_url":"https://github.com/Satan012/BERT/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4f87a173bcbfc6e5","mcp_get_code":{"code_sha256":"4f87a173bcbfc6e5"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"h4ste/oscar","path":"modeling.py","file_url":"https://github.com/h4ste/oscar/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f0fc6275fce8ef79","mcp_get_code":{"code_sha256":"f0fc6275fce8ef79"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chiayewken/bert-qa","path":"modeling.py","file_url":"https://github.com/chiayewken/bert-qa/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5e82334665dedf36","mcp_get_code":{"code_sha256":"5e82334665dedf36"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"guoyaohua/BERT-Classifier","path":"modeling.py","file_url":"https://github.com/guoyaohua/BERT-Classifier/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"60bfa1dabc7ab848","mcp_get_code":{"code_sha256":"60bfa1dabc7ab848"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pengshuyuan/Bert","path":"modeling.py","file_url":"https://github.com/pengshuyuan/Bert/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"05cc60302ae6f12e","mcp_get_code":{"code_sha256":"05cc60302ae6f12e"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zsweet/BERT_zsw","path":"modeling.py","file_url":"https://github.com/zsweet/BERT_zsw/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"95cf64fa85a7e558","mcp_get_code":{"code_sha256":"95cf64fa85a7e558"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"KnightZhang625/BERT_TF","path":"model_official.py","file_url":"https://github.com/KnightZhang625/BERT_TF/blob/HEAD/model_official.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4525f9e4207ffc06","mcp_get_code":{"code_sha256":"4525f9e4207ffc06"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deepmipt/bert","path":"bert_dp/modeling.py","file_url":"https://github.com/deepmipt/bert/blob/HEAD/bert_dp/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b9f693772b62b1de","mcp_get_code":{"code_sha256":"b9f693772b62b1de"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haidershaour/bert","path":"modeling.py","file_url":"https://github.com/haidershaour/bert/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ea7da1391dc2610d","mcp_get_code":{"code_sha256":"ea7da1391dc2610d"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yydai/bert_test","path":"modeling.py","file_url":"https://github.com/yydai/bert_test/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6665fea4d7527bd9","mcp_get_code":{"code_sha256":"6665fea4d7527bd9"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SCismycat/bert_code_view","path":"modeling.py","file_url":"https://github.com/SCismycat/bert_code_view/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a98b0db74fba302e","mcp_get_code":{"code_sha256":"a98b0db74fba302e"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"semal/bert","path":"bert/core/modeling.py","file_url":"https://github.com/semal/bert/blob/HEAD/bert/core/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c35f055903ebe93d","mcp_get_code":{"code_sha256":"c35f055903ebe93d"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MichaelZhouwang/LMlexsub","path":"modeling.py","file_url":"https://github.com/MichaelZhouwang/LMlexsub/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fa4fc5f028a8331e","mcp_get_code":{"code_sha256":"fa4fc5f028a8331e"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ricardordb/bert","path":"modeling.py","file_url":"https://github.com/ricardordb/bert/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ec6e2c37d8c8f1a7","mcp_get_code":{"code_sha256":"ec6e2c37d8c8f1a7"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"appcoreopc/berty","path":"modeling.py","file_url":"https://github.com/appcoreopc/berty/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3f7055cb9134acc8","mcp_get_code":{"code_sha256":"3f7055cb9134acc8"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JNUpython/bert","path":"bert_base/modeling.py","file_url":"https://github.com/JNUpython/bert/blob/HEAD/bert_base/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c51e39f79c51d162","mcp_get_code":{"code_sha256":"c51e39f79c51d162"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"goodluck4s/bert-","path":"modeling.py","file_url":"https://github.com/goodluck4s/bert-/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3ecf4675e826ebbd","mcp_get_code":{"code_sha256":"3ecf4675e826ebbd"}},{"arxiv_id":"1706.03762","paper":"/paper/attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"goldenbili/Bert_Test2","path":"modeling.py","file_url":"https://github.com/goldenbili/Bert_Test2/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"751674c1b265282d","mcp_get_code":{"code_sha256":"751674c1b265282d"}},{"arxiv_id":"2023.acl-long.693","paper":null,"title":"arXiv:2023.acl-long.693","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"AI4Bharat/IndicBERT","path":"train/modeling.py","file_url":"https://github.com/AI4Bharat/IndicBERT/blob/HEAD/train/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a8954909e0e65298","mcp_get_code":{"code_sha256":"a8954909e0e65298"}},{"arxiv_id":"2021.acl-long.233","paper":null,"title":"arXiv:2021.acl-long.233","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ACL2020SpellGCN/SpellGCN","path":"modeling.py","file_url":"https://github.com/ACL2020SpellGCN/SpellGCN/blob/HEAD/modeling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4b8758c7c102ebb6","mcp_get_code":{"code_sha256":"4b8758c7c102ebb6"}}]}