{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/load-balancing-loss-func","entry":"load_balancing_loss_func","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-25T09:33:49+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":24,"n_papers_ran":13,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":17,"n_samples_ran":12,"n_samples_fingerprinted":0,"n_places":27,"n_places_pointer_only":14,"by_status":{"ran_honours":1,"ran_violates":2,"ran_draft_wrong":0,"ran_fixture":0,"ran":9,"unverified":5},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.02204","paper":"/paper/arxiv-2609-02204","title":"TAME: Temporal-Aware Mixture-of-Experts for Text-Video Retrieval","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"sejong-rcv/TAME","path":"modules/MoE_utils.py","file_url":"https://github.com/sejong-rcv/TAME/blob/HEAD/modules/MoE_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3e863ec151094058","mcp_get_code":{"code_sha256":"3e863ec151094058"}},{"arxiv_id":"2609.02204","paper":"/paper/arxiv-2609-02204","title":"TAME: Temporal-Aware Mixture-of-Experts for Text-Video Retrieval","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"sejong-rcv/TAME","path":"modules/until_module.py","file_url":"https://github.com/sejong-rcv/TAME/blob/HEAD/modules/until_module.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9b9174b87f63cc4c","mcp_get_code":{"code_sha256":"9b9174b87f63cc4c"}},{"arxiv_id":"2605.30992","paper":"/paper/arxiv-2605-30992","title":"Eigenvectors of Experts are Training-free Non-collapsing Routers","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"giangdip2410/SSMoE","path":"SSMoE/Language/SSMoE-Embedding/models/modeling_olmoe.py","file_url":"https://github.com/giangdip2410/SSMoE/blob/HEAD/SSMoE/Language/SSMoE-Embedding/models/modeling_olmoe.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d0725ff0666ee959","mcp_get_code":{"code_sha256":"d0725ff0666ee959"}},{"arxiv_id":"2605.30992","paper":"/paper/arxiv-2605-30992","title":"Eigenvectors of Experts are Training-free Non-collapsing Routers","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"giangdip2410/SSMoE","path":"SSMoE/Language/SSMoE-Embedding/models/modeling_gritlm8x7b.py","file_url":"https://github.com/giangdip2410/SSMoE/blob/HEAD/SSMoE/Language/SSMoE-Embedding/models/modeling_gritlm8x7b.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e566aca3f6f9392e","mcp_get_code":{"code_sha256":"e566aca3f6f9392e"}},{"arxiv_id":"2605.30992","paper":"/paper/arxiv-2605-30992","title":"Eigenvectors of Experts are Training-free Non-collapsing Routers","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"giangdip2410/SSMoE","path":"SSMoE/Vision/CLIP-SSMoE/clipmoe/model_clipmoe.py","file_url":"https://github.com/giangdip2410/SSMoE/blob/HEAD/SSMoE/Vision/CLIP-SSMoE/clipmoe/model_clipmoe.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"25b8549e7bd56c7d","mcp_get_code":{"code_sha256":"25b8549e7bd56c7d"}},{"arxiv_id":"2602.16052","paper":"/paper/arxiv-2602-16052","title":"MoE-Spec: Expert Budgeting for Efficient Speculative Decoding","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"SafeAILab/EAGLE","path":"eagle/model/modeling_mixtral_kv.py","file_url":"https://github.com/SafeAILab/EAGLE/blob/HEAD/eagle/model/modeling_mixtral_kv.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d9308d9bf18cd072","mcp_get_code":{"code_sha256":"d9308d9bf18cd072"}},{"arxiv_id":"2601.19278","paper":"/paper/arxiv-2601-19278","title":"DART: Diffusion-Inspired Speculative Decoding for Fast LLM Inference","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"fvliang/DART","path":"dart/model/modeling_mixtral_kv.py","file_url":"https://github.com/fvliang/DART/blob/HEAD/dart/model/modeling_mixtral_kv.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d9308d9bf18cd072","mcp_get_code":{"code_sha256":"d9308d9bf18cd072"}},{"arxiv_id":"2601.15498","paper":"/paper/arxiv-2601-15498","title":"MARS: Unleashing the Power of Speculative Decoding via Margin-Aware Verification","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"5SSjw/MARS","path":"mars/model/modeling_mixtral_kv.py","file_url":"https://github.com/5SSjw/MARS/blob/HEAD/mars/model/modeling_mixtral_kv.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d9308d9bf18cd072","mcp_get_code":{"code_sha256":"d9308d9bf18cd072"}},{"arxiv_id":"2509.15235","paper":"/paper/arxiv-2509-15235","title":"ViSpec: Accelerating Vision-Language Models with Vision-Aware Speculative Decoding","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"KangJialiang/ViSpec","path":"vispec/model/modeling_mixtral_kv.py","file_url":"https://github.com/KangJialiang/ViSpec/blob/HEAD/vispec/model/modeling_mixtral_kv.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d9308d9bf18cd072","mcp_get_code":{"code_sha256":"d9308d9bf18cd072"}},{"arxiv_id":"2507.22424","paper":null,"title":"arXiv:2507.22424","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"PineTreeWss/SpecVLA","path":"openvla/specdecoding/model/modeling_mixtral_kv.py","file_url":"https://github.com/PineTreeWss/SpecVLA/blob/HEAD/openvla/specdecoding/model/modeling_mixtral_kv.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d9308d9bf18cd072","mcp_get_code":{"code_sha256":"d9308d9bf18cd072"}},{"arxiv_id":"2506.13585","paper":"/paper/minimax-m1-scaling-test-time-compute","title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention","date":"2025-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"minimax-ai/minimax-m1","path":"modeling_minimax_m1.py","file_url":"https://github.com/minimax-ai/minimax-m1/blob/HEAD/modeling_minimax_m1.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2d7fb5c16b69ee87","mcp_get_code":{"code_sha256":"2d7fb5c16b69ee87"}},{"arxiv_id":"2505.17257","paper":"/paper/janusdna-a-powerful-bi-directional-hybrid-dna","title":"JanusDNA: A Powerful Bi-directional Hybrid DNA Foundation Model","date":"2025-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Qihao-Duan/JanusDNA","path":"janusdna/modeling_janusdna.py","file_url":"https://github.com/Qihao-Duan/JanusDNA/blob/HEAD/janusdna/modeling_janusdna.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"afa4dd8b92112788","mcp_get_code":{"code_sha256":"afa4dd8b92112788"}},{"arxiv_id":"2503.06881","paper":"/paper/resmoe-space-efficient-compression-of-mixture","title":"ResMoE: Space-efficient Compression of Mixture of Experts LLMs via Residual Restoration","date":"2025-03-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"idea-isail-lab-uiuc/resmoe","path":"mixtral/resmoe_mixtral/modeling_mixtral.py","file_url":"https://github.com/idea-isail-lab-uiuc/resmoe/blob/HEAD/mixtral/resmoe_mixtral/modeling_mixtral.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e566aca3f6f9392e","mcp_get_code":{"code_sha256":"e566aca3f6f9392e"}},{"arxiv_id":"2501.04961","paper":"/paper/demystifying-domain-adaptive-post-training","title":"Demystifying Domain-adaptive Post-training for Financial LLMs","date":"2025-01-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforceairesearch/findap","path":"FinDAP/utils/packing/monkey_patch_packing.py","file_url":"https://github.com/salesforceairesearch/findap/blob/HEAD/FinDAP/utils/packing/monkey_patch_packing.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"ecadd5acaff4db07","mcp_get_code":{"code_sha256":"ecadd5acaff4db07"}},{"arxiv_id":"2410.10814","paper":"/paper/your-mixture-of-experts-llm-is-secretly-an","title":"Your Mixture-of-Experts LLM Is Secretly an Embedding Model For Free","date":"2024-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tianyi-lab/moe-embedding","path":"models/modeling_olmoe.py","file_url":"https://github.com/tianyi-lab/moe-embedding/blob/HEAD/models/modeling_olmoe.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d0725ff0666ee959","mcp_get_code":{"code_sha256":"d0725ff0666ee959"}},{"arxiv_id":"2409.16040","paper":"/paper/time-moe-billion-scale-time-series-foundation","title":"Time-MoE: Billion-Scale Time Series Foundation Models with Mixture of Experts","date":"2024-09-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Time-MoE/Time-MoE","path":"time_moe/models/modeling_time_moe.py","file_url":"https://github.com/Time-MoE/Time-MoE/blob/HEAD/time_moe/models/modeling_time_moe.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e0d3c24aaf39d620","mcp_get_code":{"code_sha256":"e0d3c24aaf39d620"}},{"arxiv_id":"2408.00264","paper":"/paper/clover-2-accurate-inference-for-regressive","title":"Clover-2: Accurate Inference for Regressive Lightweight Speculative Decoding","date":"2024-08-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"XiaoBin1992/clover","path":"clover/model/modeling_mixtral_kv.py","file_url":"https://github.com/XiaoBin1992/clover/blob/HEAD/clover/model/modeling_mixtral_kv.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d9308d9bf18cd072","mcp_get_code":{"code_sha256":"d9308d9bf18cd072"}},{"arxiv_id":"2406.19598","paper":"/paper/mixture-of-in-context-experts-enhance-llms","title":"Mixture of In-Context Experts Enhance LLMs' Long Context Awareness","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"p1nksnow/moice","path":"modeling/modeling_llama.py","file_url":"https://github.com/p1nksnow/moice/blob/HEAD/modeling/modeling_llama.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"114e672ea51890e7","mcp_get_code":{"code_sha256":"114e672ea51890e7"}},{"arxiv_id":"2406.13233","paper":"/paper/adamoe-token-adaptive-routing-with-null","title":"AdaMoE: Token-Adaptive Routing with Null Experts for Mixture-of-Experts Language Models","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cengzihao/adamoe","path":"exp_1/src/mola_modeling_llama_hacked.py","file_url":"https://github.com/cengzihao/adamoe/blob/HEAD/exp_1/src/mola_modeling_llama_hacked.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e9ff56fcac52d4ac","mcp_get_code":{"code_sha256":"e9ff56fcac52d4ac"}},{"arxiv_id":"2405.13845","paper":"/paper/semantic-density-uncertainty-quantification","title":"Semantic Density: Uncertainty Quantification for Large Language Models through Confidence Measurement in Semantic Space","date":"2024-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cognizant-ai-labs/semantic-density-paper","path":"huggingface_replacement/modeling_mixtral.py","file_url":"https://github.com/cognizant-ai-labs/semantic-density-paper/blob/HEAD/huggingface_replacement/modeling_mixtral.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d9308d9bf18cd072","mcp_get_code":{"code_sha256":"d9308d9bf18cd072"}},{"arxiv_id":"2404.15574","paper":"/paper/retrieval-head-mechanistically-explains-long","title":"Retrieval Head Mechanistically Explains Long-Context Factuality","date":"2024-04-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nightdessert/retrieval_head","path":"faiss_attn/source/modeling_mixtral.py","file_url":"https://github.com/nightdessert/retrieval_head/blob/HEAD/faiss_attn/source/modeling_mixtral.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"71b1358cf9d3dbda","mcp_get_code":{"code_sha256":"71b1358cf9d3dbda"}},{"arxiv_id":"2402.12030","paper":"/paper/towards-cross-tokenizer-distillation-the","title":"Towards Cross-Tokenizer Distillation: the Universal Logit Distillation Loss for LLMs","date":"2024-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"arcee-ai/distillkit","path":"distillkit/monkey_patch_packing.py","file_url":"https://github.com/arcee-ai/distillkit/blob/HEAD/distillkit/monkey_patch_packing.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e4ac3e44dca69ef9","mcp_get_code":{"code_sha256":"e4ac3e44dca69ef9"}},{"arxiv_id":"2402.08562","paper":"/paper/higher-layers-need-more-lora-experts","title":"Higher Layers Need More LoRA Experts","date":"2024-02-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gcyzsl/mola","path":"src/mola_modeling_llama_hacked.py","file_url":"https://github.com/gcyzsl/mola/blob/HEAD/src/mola_modeling_llama_hacked.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"94e2e2c1c3336f07","mcp_get_code":{"code_sha256":"94e2e2c1c3336f07"}},{"arxiv_id":"2310.16450","paper":"/paper/clex-continuous-length-extrapolation-for","title":"CLEX: Continuous Length Extrapolation for Large Language Models","date":"2023-10-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DAMO-NLP-SG/CLEX","path":"CLEX/mixtral/modeling_mixtral_clex.py","file_url":"https://github.com/DAMO-NLP-SG/CLEX/blob/HEAD/CLEX/mixtral/modeling_mixtral_clex.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"74f2d25c8432bf82","mcp_get_code":{"code_sha256":"74f2d25c8432bf82"}},{"arxiv_id":"2310.11511","paper":"/paper/self-rag-learning-to-retrieve-generate-and","title":"Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection","date":"2023-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ShayekhBinIslam/openrag","path":"openrag/modeling_openrag.py","file_url":"https://github.com/ShayekhBinIslam/openrag/blob/HEAD/openrag/modeling_openrag.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC-BY-4.0","inline_ok":false,"code_sha256_prefix":"d9308d9bf18cd072","mcp_get_code":{"code_sha256":"d9308d9bf18cd072"}},{"arxiv_id":"openreview_4iupzej9nT","paper":null,"title":"arXiv:openreview_4iupzej9nT","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"hikvision-research/STEP","path":"step/model/olmoe/modeling_olmoe.py","file_url":"https://github.com/hikvision-research/STEP/blob/HEAD/step/model/olmoe/modeling_olmoe.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"84f6b3be4b370b1a","mcp_get_code":{"code_sha256":"84f6b3be4b370b1a"}},{"arxiv_id":"2025.emnlp-main.231","paper":null,"title":"arXiv:2025.emnlp-main.231","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"chenziwenhaoshuai/TS-CLIP","path":"time_moe/models/modeling_time_moe.py","file_url":"https://github.com/chenziwenhaoshuai/TS-CLIP/blob/HEAD/time_moe/models/modeling_time_moe.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e0d3c24aaf39d620","mcp_get_code":{"code_sha256":"e0d3c24aaf39d620"}}]}