{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/rmsnorm","entry":"RMSNorm","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":61,"n_papers_ran":59,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":63,"n_samples_ran":61,"n_samples_fingerprinted":8,"n_places":63,"n_places_pointer_only":18,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":61,"unverified":2},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.24597","paper":"/paper/arxiv-2608-24597","title":"Taming foundation model with invariance-oriented pre-training for broad-spectrum EEG analysis across signal-level, brain-state, and brain-health tasks","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"jingyingma01/CodeBrain","path":"Models/SSSM.py","file_url":"https://github.com/jingyingma01/CodeBrain/blob/HEAD/Models/SSSM.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"eb1aca90c956ed92","mcp_get_code":{"code_sha256":"eb1aca90c956ed92"}},{"arxiv_id":"2608.07249","paper":"/paper/arxiv-2608-07249","title":"Stoicheia: Character-Level Masked Diffusion for Ancient Greek Textual Restoration, Parsing, and Metrical Scansion","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"ericu9500/stoicheia","path":"hf_release/modeling_char_bert_joint.py","file_url":"https://github.com/ericu9500/stoicheia/blob/HEAD/hf_release/modeling_char_bert_joint.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2286a7c35ee41c87","mcp_get_code":{"code_sha256":"2286a7c35ee41c87"}},{"arxiv_id":"2608.01298","paper":"/paper/arxiv-2608-01298","title":"UDT: Reconciling U-Nets and Diffusion Transformers with Data-Adaptive Token Reduction","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"JN-Yun/UDT","path":"models/UDT.py","file_url":"https://github.com/JN-Yun/UDT/blob/HEAD/models/UDT.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9a95d88a2519390e","mcp_get_code":{"code_sha256":"9a95d88a2519390e"}},{"arxiv_id":"2607.26504","paper":"/paper/arxiv-2607-26504","title":"From Interface to Inference: Eliciting Any-Order Inference from Any-Order Models","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"SeunggeunKimkr/genuine-any-order","path":"LatentMDM/model/latent_mdm.py","file_url":"https://github.com/SeunggeunKimkr/genuine-any-order/blob/HEAD/LatentMDM/model/latent_mdm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3960fafde947a0b8","mcp_get_code":{"code_sha256":"3960fafde947a0b8"}},{"arxiv_id":"2607.18290","paper":"/paper/arxiv-2607-18290","title":"SechKAN: Kolmogorov-Arnold Networks with Hyperbolic Secant Functions","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"hoangthangta/All-KAN","path":"models/sech_kan.py","file_url":"https://github.com/hoangthangta/All-KAN/blob/HEAD/models/sech_kan.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1d4a7d068cf45d10","mcp_get_code":{"code_sha256":"1d4a7d068cf45d10"}},{"arxiv_id":"2606.22627","paper":"/paper/arxiv-2606-22627","title":"Orthogonal Representation Editing: Decoupling Semantic Entanglement in Batch Knowledge Editing of LLMs","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"YVVH/ORE","path":"ORE/ORE_main.py","file_url":"https://github.com/YVVH/ORE/blob/HEAD/ORE/ORE_main.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6347bf161968fa04","mcp_get_code":{"code_sha256":"6347bf161968fa04"}},{"arxiv_id":"2606.22248","paper":"/paper/arxiv-2606-22248","title":"SamatNext v0.2-B: An Exploratory Study of RMS-Normalized Hybrid Decoders for Curriculum Retention in Small Code Models","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"samat2003/samatnext-v0.1","path":"models/samat_next/model.py","file_url":"https://github.com/samat2003/samatnext-v0.1/blob/HEAD/models/samat_next/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"561893a8e22a4976","mcp_get_code":{"code_sha256":"561893a8e22a4976"}},{"arxiv_id":"2606.21973","paper":"/paper/arxiv-2606-21973","title":"SPOTR: Spatio-temporal Pooling One-Token Reconstruction for Universal Physiological Signal Self-supervised Learning","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"5GYYYYY/SPOTR","path":"SPOTR.py","file_url":"https://github.com/5GYYYYY/SPOTR/blob/HEAD/SPOTR.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5640196b424b0268","mcp_get_code":{"code_sha256":"5640196b424b0268"}},{"arxiv_id":"2606.21911","paper":"/paper/arxiv-2606-21911","title":"The Pitfall of Scaling Up: Uncovering and Mitigating Popularity Bias Amplification in Scaling Transformer-based Recommenders","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Tiny-Snow/GenRec","path":"src/genrec/models/model_seqrec/sasrec_sprint.py","file_url":"https://github.com/Tiny-Snow/GenRec/blob/HEAD/src/genrec/models/model_seqrec/sasrec_sprint.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"82a2d670a351cbcf","mcp_get_code":{"code_sha256":"82a2d670a351cbcf"}},{"arxiv_id":"2606.08414","paper":"/paper/arxiv-2606-08414","title":"PACT: Self-Evolving Physical Safety Alignment for Diffusion Policies in Embodied Manipulation","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"thu-ml/RDT2","path":"models/rdt/model.py","file_url":"https://github.com/thu-ml/RDT2/blob/HEAD/models/rdt/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e1daa20aa3ecf380","mcp_get_code":{"code_sha256":"e1daa20aa3ecf380"}},{"arxiv_id":"2606.01495","paper":"/paper/arxiv-2606-01495","title":"CART: Context-Anchored Recurrent Transformer * A Parameter-Efficient Architecture with Learned Stability","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"ccapps42/CART","path":"model/cart.py","file_url":"https://github.com/ccapps42/CART/blob/HEAD/model/cart.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5e72e01b3a174cc0","mcp_get_code":{"code_sha256":"5e72e01b3a174cc0"}},{"arxiv_id":"2605.15488","paper":"/paper/arxiv-2605-15488","title":"SurvivalPFN: Amortizing Survival Prediction via In-Context Bayesian Inference","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"rgklab/SurvivalPFN","path":"survivalpfn/models/icl_model.py","file_url":"https://github.com/rgklab/SurvivalPFN/blob/HEAD/survivalpfn/models/icl_model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6e57086fc97463e2","mcp_get_code":{"code_sha256":"6e57086fc97463e2"}},{"arxiv_id":"2605.15133","paper":"/paper/arxiv-2605-15133","title":"Causal Foundation Models with Continuous Treatments","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"layer6ai-labs/CCPFN-inference","path":"src/ccpfn/models/continuous_icl_model.py","file_url":"https://github.com/layer6ai-labs/CCPFN-inference/blob/HEAD/src/ccpfn/models/continuous_icl_model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8ebbbd4e10f29278","mcp_get_code":{"code_sha256":"8ebbbd4e10f29278"}},{"arxiv_id":"2605.13989","paper":"/paper/arxiv-2605-13989","title":"VectraYX-Nano: A 42M-Parameter Spanish Cybersecurity Language Model with Curriculum Learning and Native Tool Use","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"vectrayx/vectrayx-nano-paper","path":"training/transformer.py","file_url":"https://github.com/vectrayx/vectrayx-nano-paper/blob/HEAD/training/transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"05a42e70386b7c33","mcp_get_code":{"code_sha256":"05a42e70386b7c33"}},{"arxiv_id":"2605.09742","paper":"/paper/arxiv-2605-09742","title":"TIDES: Implicit Time-Awareness in Selective State Space Models","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"TaylanSoydan/TIDES","path":"tides/tides.py","file_url":"https://github.com/TaylanSoydan/TIDES/blob/HEAD/tides/tides.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ae412fb79bbc7983","mcp_get_code":{"code_sha256":"ae412fb79bbc7983"}},{"arxiv_id":"2604.22826","paper":"/paper/arxiv-2604-22826","title":"Shape: A Self-Supervised 3D Geometry Foundation Model for Industrial CAD Analysis","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"simd-ai/shape","path":"shape_foundation/models/gaot_backbone.py","file_url":"https://github.com/simd-ai/shape/blob/HEAD/shape_foundation/models/gaot_backbone.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7b2150d02885a569","mcp_get_code":{"code_sha256":"7b2150d02885a569"}},{"arxiv_id":"2604.21649","paper":"/paper/arxiv-2604-21649","title":"GS-Quant: Granular Semantic and Generative Structural Quantization for Knowledge Graph Completion","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"mikumifa/GS-Quant","path":"codebook/rqvae.py","file_url":"https://github.com/mikumifa/GS-Quant/blob/HEAD/codebook/rqvae.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e90fe98be09c54bd","mcp_get_code":{"code_sha256":"e90fe98be09c54bd"}},{"arxiv_id":"2603.14366","paper":"/paper/arxiv-2603-14366","title":"Representation Alignment for Just Image Transformers is not Easier than You Think","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"kaist-cvml/PixelREPA","path":"model_pixelREPA.py","file_url":"https://github.com/kaist-cvml/PixelREPA/blob/HEAD/model_pixelREPA.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"77d34b3ccaee6e17","mcp_get_code":{"code_sha256":"77d34b3ccaee6e17"}},{"arxiv_id":"2603.07020","paper":"/paper/arxiv-2603-07020","title":"RESCHED: Rethinking Flexible Job Shop Scheduling from a Transformer-based Architecture with Simplified States","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"XiangjieXiao/ReSched","path":"PPO/SchedulingModel.py","file_url":"https://github.com/XiangjieXiao/ReSched/blob/HEAD/PPO/SchedulingModel.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9465fd5c593cbef5","mcp_get_code":{"code_sha256":"9465fd5c593cbef5"}},{"arxiv_id":"2602.22286","paper":"/paper/arxiv-2602-22286","title":"OmniZip: Learning a Unified and Lightweight Lossless Compressor for Multi-Modal Data","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"adminasmi/OmniZip-CVPR2026","path":"models/rwkv7_hira_moa_moe.py","file_url":"https://github.com/adminasmi/OmniZip-CVPR2026/blob/HEAD/models/rwkv7_hira_moa_moe.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c1ff30fd8e32c15e","mcp_get_code":{"code_sha256":"c1ff30fd8e32c15e"}},{"arxiv_id":"2602.14615","paper":"/paper/arxiv-2602-14615","title":"VariViT: A Vision Transformer for Variable Image Sizes","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"Aswathi-Varma/varivit","path":"model/navit.py","file_url":"https://github.com/Aswathi-Varma/varivit/blob/HEAD/model/navit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"02d91e9f7f403be0","mcp_get_code":{"code_sha256":"02d91e9f7f403be0"}},{"arxiv_id":"2602.02493","paper":"/paper/arxiv-2602-02493","title":"PixelGen: Improving Pixel Diffusion with Perceptual Supervision","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"Zehong-Ma/PixelGen","path":"src/models/transformer/JiT.py","file_url":"https://github.com/Zehong-Ma/PixelGen/blob/HEAD/src/models/transformer/JiT.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"84963760feac9e7d","mcp_get_code":{"code_sha256":"84963760feac9e7d"}},{"arxiv_id":"2601.21579","paper":"/paper/arxiv-2601-21579","title":"KromHC: Manifold-Constrained Hyper-Connections with Kronecker-Product Residual Matrices","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"wz1119/KromHC","path":"hyper_conn/Kromhc.py","file_url":"https://github.com/wz1119/KromHC/blob/HEAD/hyper_conn/Kromhc.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9570166c272034c8","mcp_get_code":{"code_sha256":"9570166c272034c8"}},{"arxiv_id":"2601.01313","paper":"/paper/arxiv-2601-01313","title":"Spectral-Window Hybrid (SWH)","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"VladimerKhasia/SWH","path":"swh.py","file_url":"https://github.com/VladimerKhasia/SWH/blob/HEAD/swh.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"132d910895b3fe88","mcp_get_code":{"code_sha256":"132d910895b3fe88"}},{"arxiv_id":"2512.20251","paper":"/paper/arxiv-2512-20251","title":"Degradation-Aware Metric Prompting for Hyperspectral Image Restoration","date":null,"month_inferred_from_arxiv_id":"2025-12","title_source":"syntology","repo":"MiliLab/DAMP","path":"DAMP.py","file_url":"https://github.com/MiliLab/DAMP/blob/HEAD/DAMP.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6aba4598e0003aae","mcp_get_code":{"code_sha256":"6aba4598e0003aae"}},{"arxiv_id":"2511.00833","paper":"/paper/arxiv-2511-00833","title":"Linear Differential Vision Transformer: Learning Visual Contrasts via Pairwise Differentials","date":null,"month_inferred_from_arxiv_id":"2025-11","title_source":"syntology","repo":"LeapLabTHU/LinearDiff","path":"image_classification/models/vca_deit.py","file_url":"https://github.com/LeapLabTHU/LinearDiff/blob/HEAD/image_classification/models/vca_deit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"af5816548ff50622","mcp_get_code":{"code_sha256":"af5816548ff50622"}},{"arxiv_id":"2510.19710","paper":"/paper/arxiv-2510-19710","title":"SEMPO: Lightweight Foundation Models for Time Series Forecasting","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"mala-lab/SEMPO","path":"models/SEMPO.py","file_url":"https://github.com/mala-lab/SEMPO/blob/HEAD/models/SEMPO.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0a5b83bc7ce0cd66","mcp_get_code":{"code_sha256":"0a5b83bc7ce0cd66"}},{"arxiv_id":"2510.04577","paper":"/paper/arxiv-2510-04577","title":"Language Model Based Text-to-Audio Generation: Anti-Causally Aligned Collaborative Residual Transformers","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"wjc2830/Siren","path":"src/siren/model.py","file_url":"https://github.com/wjc2830/Siren/blob/HEAD/src/siren/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9669b074ff77501a","mcp_get_code":{"code_sha256":"9669b074ff77501a"}},{"arxiv_id":"2509.14630","paper":"/paper/arxiv-2509-14630","title":"Toward Embodiment Equivariant Vision-Language-Action Policy","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"hhcaz/e2vla","path":"models/action_expert.py","file_url":"https://github.com/hhcaz/e2vla/blob/HEAD/models/action_expert.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ea08af0febc2e0f0","mcp_get_code":{"code_sha256":"ea08af0febc2e0f0"}},{"arxiv_id":"2507.06363","paper":null,"title":"arXiv:2507.06363","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"gmum/MambaHoME","path":"src/models/soft_moe_2d_3d.py","file_url":"https://github.com/gmum/MambaHoME/blob/HEAD/src/models/soft_moe_2d_3d.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5e5be82456aab234","mcp_get_code":{"code_sha256":"5e5be82456aab234"}},{"arxiv_id":"2506.19935","paper":"/paper/any-order-gpt-as-masked-diffusion-model","title":"Any-Order GPT as Masked Diffusion Model: Decoupling Formulation and Architecture","date":"2025-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"scxue/AO-GPT-MDM","path":"model_AOGPT_AdaLN6_NoRep_cond_128_trunc_qknorm.py","file_url":"https://github.com/scxue/AO-GPT-MDM/blob/HEAD/model_AOGPT_AdaLN6_NoRep_cond_128_trunc_qknorm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7d84ead0fe5b239f","mcp_get_code":{"code_sha256":"7d84ead0fe5b239f"}},{"arxiv_id":"2505.02707","paper":"/paper/voila-voice-language-foundation-models-for","title":"Voila: Voice-Language Foundation Models for Real-Time Autonomous Interaction and Voice Role-Play","date":"2025-05-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"maitrix-org/Voila","path":"model.py","file_url":"https://github.com/maitrix-org/Voila/blob/HEAD/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"25e14d2b6e56efe1","mcp_get_code":{"code_sha256":"25e14d2b6e56efe1"}},{"arxiv_id":"2504.07963","paper":"/paper/pixelflow-pixel-space-generative-models-with","title":"PixelFlow: Pixel-Space Generative Models with Flow","date":"2025-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shoufachen/pixelflow","path":"pixelflow/model.py","file_url":"https://github.com/shoufachen/pixelflow/blob/HEAD/pixelflow/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a5e686c656d9d355","mcp_get_code":{"code_sha256":"a5e686c656d9d355"}},{"arxiv_id":"2502.17363","paper":"/paper/kv-edit-training-free-image-editing-for","title":"KV-Edit: Training-Free Image Editing for Precise Background Preservation","date":"2025-02-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Xilluill/KV-Edit","path":"models/kv_edit.py","file_url":"https://github.com/Xilluill/KV-Edit/blob/HEAD/models/kv_edit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b4affeaaac4bcdd4","mcp_get_code":{"code_sha256":"b4affeaaac4bcdd4"}},{"arxiv_id":"2502.07244","paper":"/paper/linear-transformers-as-var-models-aligning","title":"Linear Transformers as VAR Models: Aligning Autoregressive Attention Mechanisms with Autoregressive Forecasting","date":"2025-02-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ljc-fvnr/structural-aligned-mixture-of-var","path":"models/AutoregressiveAlignment.py","file_url":"https://github.com/ljc-fvnr/structural-aligned-mixture-of-var/blob/HEAD/models/AutoregressiveAlignment.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"45502bf8690c09af","mcp_get_code":{"code_sha256":"45502bf8690c09af"}},{"arxiv_id":"2502.04320","paper":"/paper/conceptattention-diffusion-transformers-learn","title":"ConceptAttention: Diffusion Transformers Learn Highly Interpretable Features","date":"2025-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"helblazer811/ConceptAttention","path":"concept_attention/flux/dit_block.py","file_url":"https://github.com/helblazer811/ConceptAttention/blob/HEAD/concept_attention/flux/dit_block.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5c52ab157dc1c15c","mcp_get_code":{"code_sha256":"5c52ab157dc1c15c"}},{"arxiv_id":"2502.01105","paper":"/paper/layertracer-cognitive-aligned-layered-svg","title":"LayerTracer: Cognitive-Aligned Layered SVG Synthesis via Diffusion Transformer","date":"2025-02-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"showlab/LayerTracer","path":"library/flux_models.py","file_url":"https://github.com/showlab/LayerTracer/blob/HEAD/library/flux_models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4de3b3072dadcd2f","mcp_get_code":{"code_sha256":"4de3b3072dadcd2f"}},{"arxiv_id":"2501.15461","paper":"/paper/mamba-based-graph-convolutional-networks","title":"Mamba-Based Graph Convolutional Networks: Tackling Over-smoothing with Selective State Space","date":"2025-01-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hexin5515/MbaGCN","path":"NodeClassification/models.py","file_url":"https://github.com/hexin5515/MbaGCN/blob/HEAD/NodeClassification/models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f79868a941682af4","mcp_get_code":{"code_sha256":"f79868a941682af4"}},{"arxiv_id":"2412.03603","paper":"/paper/hunyuanvideo-a-systematic-framework-for-large","title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","date":"2024-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tencent/hunyuanvideo","path":"hyvideo/modules/models.py","file_url":"https://github.com/tencent/hunyuanvideo/blob/HEAD/hyvideo/modules/models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"9fa452fc273f7d2a","mcp_get_code":{"code_sha256":"9fa452fc273f7d2a"}},{"arxiv_id":"2410.10356","paper":"/paper/fasterdit-towards-faster-diffusion","title":"FasterDiT: Towards Faster Diffusion Transformers Training without Architecture Modification","date":"2024-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hustvl/LightningDiT","path":"models/lightningdit.py","file_url":"https://github.com/hustvl/LightningDiT/blob/HEAD/models/lightningdit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3777a235d6ea1491","mcp_get_code":{"code_sha256":"3777a235d6ea1491"}},{"arxiv_id":"2410.05258","paper":"/paper/differential-transformer","title":"Differential Transformer","date":"2024-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hammoudhasan/diffclip","path":"diff_attention.py","file_url":"https://github.com/hammoudhasan/diffclip/blob/HEAD/diff_attention.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"84b5e228ae8799e1","mcp_get_code":{"code_sha256":"84b5e228ae8799e1"}},{"arxiv_id":"2410.02705","paper":"/paper/controlar-controllable-image-generation-with","title":"ControlAR: Controllable Image Generation with Autoregressive Models","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hustvl/controlar","path":"autoregressive/models/gpt.py","file_url":"https://github.com/hustvl/controlar/blob/HEAD/autoregressive/models/gpt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"edd6b8543cc0d1ab","mcp_get_code":{"code_sha256":"edd6b8543cc0d1ab"}},{"arxiv_id":"2409.09016","paper":"/paper/closed-loop-visuomotor-control-with","title":"Closed-Loop Visuomotor Control with Generative Expectation for Robotic Manipulation","date":null,"month_inferred_from_arxiv_id":"2024-09","title_source":"archive","repo":"OpenDriveLab/CLOVER","path":"FeedbackPolicy/models/policy.py","file_url":"https://github.com/OpenDriveLab/CLOVER/blob/HEAD/FeedbackPolicy/models/policy.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"008a90cb086e88c6","mcp_get_code":{"code_sha256":"008a90cb086e88c6"}},{"arxiv_id":"2407.16448","paper":"/paper/monowad-weather-adaptive-diffusion-model-for","title":"MonoWAD: Weather-Adaptive Diffusion Model for Robust Monocular 3D Object Detection","date":"2024-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"VisualAIKHU/MonoWAD","path":"visualDet3D/networks/detectors/MonoWAD.py","file_url":"https://github.com/VisualAIKHU/MonoWAD/blob/HEAD/visualDet3D/networks/detectors/MonoWAD.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"234fd6dd12768c7d","mcp_get_code":{"code_sha256":"234fd6dd12768c7d"}},{"arxiv_id":"2407.16171","paper":"/paper/learning-trimodal-relation-for-avqa-with","title":"Learning Trimodal Relation for AVQA with Missing Modality","date":"2024-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"VisualAIKHU/Missing-AVQA","path":"net_grd_avst/net_avst.py","file_url":"https://github.com/VisualAIKHU/Missing-AVQA/blob/HEAD/net_grd_avst/net_avst.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"30e7fae83920ab01","mcp_get_code":{"code_sha256":"30e7fae83920ab01"}},{"arxiv_id":"2406.04329","paper":"/paper/simplified-and-generalized-masked-diffusion","title":"Simplified and Generalized Masked Diffusion for Discrete Data","date":"2024-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"google-deepmind/md4","path":"md4/models/diffusion/md4.py","file_url":"https://github.com/google-deepmind/md4/blob/HEAD/md4/models/diffusion/md4.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a75e172eea6d85f1","mcp_get_code":{"code_sha256":"a75e172eea6d85f1"}},{"arxiv_id":"2405.16727","paper":"/paper/disentangling-and-integrating-relational-and","title":"Disentangling and Integrating Relational and Sensory Information in Transformer Architectures","date":"2024-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"awni00/abstract_transformer","path":"dual_attention_transformer.py","file_url":"https://github.com/awni00/abstract_transformer/blob/HEAD/dual_attention_transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"01777e072296c0bb","mcp_get_code":{"code_sha256":"01777e072296c0bb"}},{"arxiv_id":"2405.13911","paper":"/paper/topa-extend-large-language-models-for-video","title":"TOPA: Extending Large Language Models for Video Understanding via Text-Only Pre-Alignment","date":"2024-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dhg-wei/topa","path":"llama/model.py","file_url":"https://github.com/dhg-wei/topa/blob/HEAD/llama/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"df852b6b3526e46e","mcp_get_code":{"code_sha256":"df852b6b3526e46e"}},{"arxiv_id":"2402.14905","paper":"/paper/mobilellm-optimizing-sub-billion-parameter","title":"MobileLLM: Optimizing Sub-billion Parameter Language Models for On-Device Use Cases","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jingyaogong/minimind","path":"model/model_minimind.py","file_url":"https://github.com/jingyaogong/minimind/blob/HEAD/model/model_minimind.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a8a84e7ddb0551b0","mcp_get_code":{"code_sha256":"a8a84e7ddb0551b0"}},{"arxiv_id":"2402.02104","paper":"/paper/learning-structure-aware-representations-of","title":"Learning Structure-Aware Representations of Dependent Types","date":"2024-02-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"konstantinoskokos/quill","path":"src/quill/nn/model.py","file_url":"https://github.com/konstantinoskokos/quill/blob/HEAD/src/quill/nn/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8183f36aee184817","mcp_get_code":{"code_sha256":"8183f36aee184817"}},{"arxiv_id":"2312.00752","paper":"/paper/mamba-linear-time-sequence-modeling-with","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","date":"2023-12-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"human9000/nd-mamba2-torch","path":"torchnssd/nd_mamba2.py","file_url":"https://github.com/human9000/nd-mamba2-torch/blob/HEAD/torchnssd/nd_mamba2.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ba026d72b685bee1","mcp_get_code":{"code_sha256":"ba026d72b685bee1"}},{"arxiv_id":"2308.00951","paper":"/paper/from-sparse-to-soft-mixtures-of-experts","title":"From Sparse to Soft Mixtures of Experts","date":"2023-08-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lucidrains/soft-moe-pytorch","path":"soft_moe_pytorch/soft_moe.py","file_url":"https://github.com/lucidrains/soft-moe-pytorch/blob/HEAD/soft_moe_pytorch/soft_moe.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"024a5f1d96cb0afb","mcp_get_code":{"code_sha256":"024a5f1d96cb0afb"}},{"arxiv_id":"2307.09288","paper":"/paper/llama-2-open-foundation-and-fine-tuned-chat","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","date":"2023-07-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Lightning-AI/lit-gpt","path":"litgpt/model.py","file_url":"https://github.com/Lightning-AI/lit-gpt/blob/HEAD/litgpt/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1e7100681c300ca2","mcp_get_code":{"code_sha256":"1e7100681c300ca2"}},{"arxiv_id":"2302.13971","paper":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"akanyaani/miniLLAMA","path":"model.py","file_url":"https://github.com/akanyaani/miniLLAMA/blob/HEAD/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ea811a4ad9027b8","mcp_get_code":{"code_sha256":"0ea811a4ad9027b8"}},{"arxiv_id":"2302.13971","paper":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"young-geng/easylm","path":"EasyLM/models/llama/llama_model.py","file_url":"https://github.com/young-geng/easylm/blob/HEAD/EasyLM/models/llama/llama_model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"aaa8a447973c0f68","mcp_get_code":{"code_sha256":"aaa8a447973c0f68"}},{"arxiv_id":"2302.13971","paper":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Lightning-AI/lit-llama","path":"lit_llama/model.py","file_url":"https://github.com/Lightning-AI/lit-llama/blob/HEAD/lit_llama/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"133fab2290f71687","mcp_get_code":{"code_sha256":"133fab2290f71687"}},{"arxiv_id":"2203.15556","paper":"/paper/training-compute-optimal-large-language","title":"Training Compute-Optimal Large Language Models","date":"2022-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"karpathy/llama2.c","path":"model.py","file_url":"https://github.com/karpathy/llama2.c/blob/HEAD/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e27b34a3f741d4de","mcp_get_code":{"code_sha256":"e27b34a3f741d4de"}},{"arxiv_id":"2202.04200","paper":"/paper/maskgit-masked-generative-image-transformer","title":"MaskGIT: Masked Generative Image Transformer","date":"2022-02-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"valeoai/maskgit-pytorch","path":"Network/transformer.py","file_url":"https://github.com/valeoai/maskgit-pytorch/blob/HEAD/Network/transformer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"86a159a05ae09375","mcp_get_code":{"code_sha256":"86a159a05ae09375"}},{"arxiv_id":"2112.04426","paper":"/paper/improving-language-models-by-retrieving-from","title":"Improving language models by retrieving from trillions of tokens","date":"2021-12-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lucidrains/RETRO-pytorch","path":"retro_pytorch/retro_pytorch.py","file_url":"https://github.com/lucidrains/RETRO-pytorch/blob/HEAD/retro_pytorch/retro_pytorch.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a4ae5f96225b96f3","mcp_get_code":{"code_sha256":"a4ae5f96225b96f3"}},{"arxiv_id":"2104.07012","paper":"/paper/sparse-attention-with-linear-units","title":"Sparse Attention with Linear Units","date":"2021-04-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rishikksh20/rectified-linear-attention","path":"attention.py","file_url":"https://github.com/rishikksh20/rectified-linear-attention/blob/HEAD/attention.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cc210ef49adb120a","mcp_get_code":{"code_sha256":"cc210ef49adb120a"}},{"arxiv_id":"2006.11239","paper":"/paper/denoising-diffusion-probabilistic-models","title":"Denoising Diffusion Probabilistic Models","date":"2020-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lucidrains/make-a-video-pytorch","path":"make_a_video_pytorch/make_a_video.py","file_url":"https://github.com/lucidrains/make-a-video-pytorch/blob/HEAD/make_a_video_pytorch/make_a_video.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7b794fd52131431f","mcp_get_code":{"code_sha256":"7b794fd52131431f"}},{"arxiv_id":"1910.07467","paper":"/paper/root-mean-square-layer-normalization","title":"Root Mean Square Layer Normalization","date":"2019-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hazdzz/RMSNorm","path":"norm.py","file_url":"https://github.com/hazdzz/RMSNorm/blob/HEAD/norm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9c235f0c904962f7","mcp_get_code":{"code_sha256":"9c235f0c904962f7"}},{"arxiv_id":"1909.08053","paper":"/paper/megatron-lm-training-multi-billion-parameter","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","date":"2019-09-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kingoflolz/mesh-transformer-jax","path":"mesh_transformer/transformer_shard.py","file_url":"https://github.com/kingoflolz/mesh-transformer-jax/blob/HEAD/mesh_transformer/transformer_shard.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ae952da8d1f3ecfc","mcp_get_code":{"code_sha256":"ae952da8d1f3ecfc"}}]}