{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/relpartiallearnablemultiheadattn","entry":"RelPartialLearnableMultiHeadAttn","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":5,"n_papers_ran":1,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":8,"n_samples_ran":1,"n_samples_fingerprinted":0,"n_places":8,"n_places_pointer_only":2,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":1,"unverified":7},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.15562","paper":"/paper/arxiv-2605-15562","title":"GiLT: Augmenting Transformer Language Models with Dependency Graphs","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"cookie-pie-oops/GiLT-LM","path":"src/model_bllip_dep.py","file_url":"https://github.com/cookie-pie-oops/GiLT-LM/blob/HEAD/src/model_bllip_dep.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3b7d8fabefd13396","mcp_get_code":{"code_sha256":"3b7d8fabefd13396"}},{"arxiv_id":"2207.06881","paper":"/paper/recurrent-memory-transformer","title":"Recurrent Memory Transformer","date":"2022-07-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"booydar/lm-rmt","path":"pytorch/mem_transformer.py","file_url":"https://github.com/booydar/lm-rmt/blob/HEAD/pytorch/mem_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"027c122b94fbe192","mcp_get_code":{"code_sha256":"027c122b94fbe192"}},{"arxiv_id":"2101.02402","paper":"/paper/compound-word-transformer-learning-to-compose","title":"Compound Word Transformer: Learning to Compose Full-Song Music over Dynamic Directed Hypergraphs","date":"2021-01-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuer867/emo-disentanger","path":"stage1_compose/model/optimus_txl_decoder.py","file_url":"https://github.com/yuer867/emo-disentanger/blob/HEAD/stage1_compose/model/optimus_txl_decoder.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8a40eebd3f2277dd","mcp_get_code":{"code_sha256":"8a40eebd3f2277dd"}},{"arxiv_id":"2004.08178","paper":"/paper/highway-transformer-self-gating-enhanced-self","title":"Highway Transformer: Self-Gating Enhanced Self-Attentive Networks","date":"2020-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cyk1337/Highway-Transformer","path":"pytorch/mem_transformer.py","file_url":"https://github.com/cyk1337/Highway-Transformer/blob/HEAD/pytorch/mem_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b232620a61168335","mcp_get_code":{"code_sha256":"b232620a61168335"}},{"arxiv_id":"1901.02860","paper":"/paper/transformer-xl-attentive-language-models","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","date":"2019-01-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shanghai-digital-brain-laboratory/bdm-db1","path":"src/model/transformer_xl.py","file_url":"https://github.com/shanghai-digital-brain-laboratory/bdm-db1/blob/HEAD/src/model/transformer_xl.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fdc345d0da45dd95","mcp_get_code":{"code_sha256":"fdc345d0da45dd95"}},{"arxiv_id":"1901.02860","paper":"/paper/transformer-xl-attentive-language-models","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","date":"2019-01-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cedrickchee/pytorch-pretrained-BERT","path":"pytorch_pretrained_bert/modeling_transfo_xl.py","file_url":"https://github.com/cedrickchee/pytorch-pretrained-BERT/blob/HEAD/pytorch_pretrained_bert/modeling_transfo_xl.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8361fa24d3e06e13","mcp_get_code":{"code_sha256":"8361fa24d3e06e13"}},{"arxiv_id":"1901.02860","paper":"/paper/transformer-xl-attentive-language-models","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","date":"2019-01-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kimiyoung/transformer-xl","path":"pytorch/mem_transformer.py","file_url":"https://github.com/kimiyoung/transformer-xl/blob/HEAD/pytorch/mem_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fb613dd5bc21371c","mcp_get_code":{"code_sha256":"fb613dd5bc21371c"}},{"arxiv_id":"1901.02860","paper":"/paper/transformer-xl-attentive-language-models","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","date":"2019-01-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"NVIDIA/DeepLearningExamples","path":"PyTorch/LanguageModeling/Transformer-XL/pytorch/mem_transformer.py","file_url":"https://github.com/NVIDIA/DeepLearningExamples/blob/HEAD/PyTorch/LanguageModeling/Transformer-XL/pytorch/mem_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5e1b71bde76b69e7","mcp_get_code":{"code_sha256":"5e1b71bde76b69e7"}}]}