{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/compute-policy-loss","entry":"compute_policy_loss","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-25T09:33:49+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":6,"n_papers_ran":4,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":6,"n_samples_ran":4,"n_samples_fingerprinted":0,"n_places":6,"n_places_pointer_only":2,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":3,"ran":0,"unverified":2},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2602.21492","paper":"/paper/arxiv-2602-21492","title":"GradAlign: Gradient-Aligned Data Selection for LLM Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"StigLidu/GradAlign","path":"select/grpo_loss.py","file_url":"https://github.com/StigLidu/GradAlign/blob/HEAD/select/grpo_loss.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9d79168f0a9a110a","mcp_get_code":{"code_sha256":"9d79168f0a9a110a"}},{"arxiv_id":"2602.13235","paper":"/paper/arxiv-2602-13235","title":"Lang2Act: Fine-Grained Visual Reasoning through Self-Emergent Linguistic Toolchains","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"NEUIR/Lang2Act","path":"src/Easyr1/verl/trainer/core_algos.py","file_url":"https://github.com/NEUIR/Lang2Act/blob/HEAD/src/Easyr1/verl/trainer/core_algos.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c4c4df4310b9d0a0","mcp_get_code":{"code_sha256":"c4c4df4310b9d0a0"}},{"arxiv_id":"2601.06021","paper":"/paper/arxiv-2601-06021","title":"Chaining the Evidence: Robust Reinforcement Learning for Deep Search Agents with Citation-Aware Rubric Rewards","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"THUDM/CaRR","path":"slime/slime/utils/ppo_utils.py","file_url":"https://github.com/THUDM/CaRR/blob/HEAD/slime/slime/utils/ppo_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43ae24183de07247","mcp_get_code":{"code_sha256":"43ae24183de07247"}},{"arxiv_id":"2510.09285","paper":"/paper/arxiv-2510-09285","title":"Spotlight on Token Perception for Multimodal Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"huaixuheqing/VPPO-RL","path":"verl/trainer/core_algos.py","file_url":"https://github.com/huaixuheqing/VPPO-RL/blob/HEAD/verl/trainer/core_algos.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0af35c786c7268c9","mcp_get_code":{"code_sha256":"0af35c786c7268c9"}},{"arxiv_id":"2506.03136","paper":"/paper/co-evolving-llm-coder-and-unit-tester-via","title":"Co-Evolving LLM Coder and Unit Tester via Reinforcement Learning","date":"2025-06-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gen-verse/reasonflux","path":"ReasonFlux_PRM/Application/verl/trainer/ppo/core_algos.py","file_url":"https://github.com/gen-verse/reasonflux/blob/HEAD/ReasonFlux_PRM/Application/verl/trainer/ppo/core_algos.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2e7d046b997dddca","mcp_get_code":{"code_sha256":"2e7d046b997dddca"}},{"arxiv_id":"2505.22453","paper":"/paper/unsupervised-post-training-for-multi-modal","title":"Unsupervised Post-Training for Multi-Modal LLM Reasoning via GRPO","date":"2025-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"waltonfuture/mm-upt","path":"verl/trainer/core_algos.py","file_url":"https://github.com/waltonfuture/mm-upt/blob/HEAD/verl/trainer/core_algos.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6650b418e4dcbad6","mcp_get_code":{"code_sha256":"6650b418e4dcbad6"}}]}