{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/get-action","entry":"get_action","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":25,"n_papers_ran":7,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":22,"n_samples_ran":8,"n_samples_fingerprinted":3,"n_places":27,"n_places_pointer_only":10,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":2,"ran":6,"unverified":14},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.09638","paper":"/paper/arxiv-2605-09638","title":"Plan2Cleanse: Test-Time Backdoor Defense via Monte-Carlo Planning in Deep Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"rl-bandits-lab/RL-Backdoor","path":"backdoor_attack/mobile_env/collect_trajectories/collect_benign_trajectories.py","file_url":"https://github.com/rl-bandits-lab/RL-Backdoor/blob/HEAD/backdoor_attack/mobile_env/collect_trajectories/collect_benign_trajectories.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"f0cfa35175a39f3e","mcp_get_code":{"code_sha256":"f0cfa35175a39f3e"}},{"arxiv_id":"2602.16855","paper":"/paper/arxiv-2602-16855","title":"Mobile-Agent-v3.5: Multi-platform Fundamental GUI Agents","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"X-PLUG/MobileAgent","path":"Mobile-Agent-v1/MobileAgent/api_service.py","file_url":"https://github.com/X-PLUG/MobileAgent/blob/HEAD/Mobile-Agent-v1/MobileAgent/api_service.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5434906159fc3994","mcp_get_code":{"code_sha256":"5434906159fc3994"}},{"arxiv_id":"2412.11499","paper":"/paper/embodied-cot-distillation-from-llm-to-off-the","title":"Embodied CoT Distillation From LLM To Off-the-shelf Agents","date":"2024-12-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"osu-nlp-group/llm-planner","path":"e2e/alfred/env/reward.py","file_url":"https://github.com/osu-nlp-group/llm-planner/blob/HEAD/e2e/alfred/env/reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"12ba53fedf724ead","mcp_get_code":{"code_sha256":"12ba53fedf724ead"}},{"arxiv_id":"2405.14751","paper":"/paper/agile-a-novel-framework-of-llm-agents","title":"AGILE: A Novel Reinforcement Learning Framework of LLM Agents","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytarnish/AGILE","path":"agile/hotpot_agent.py","file_url":"https://github.com/bytarnish/AGILE/blob/HEAD/agile/hotpot_agent.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4f65da0e3cfe9f59","mcp_get_code":{"code_sha256":"4f65da0e3cfe9f59"}},{"arxiv_id":"2403.09859","paper":"/paper/mamba-an-effective-world-model-approach-for","title":"MAMBA: an Effective World Model Approach for Meta-Reinforcement Learning","date":"2024-03-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zoharri/mamba","path":"policy_inference.py","file_url":"https://github.com/zoharri/mamba/blob/HEAD/policy_inference.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"4700cde0d0e8a52a","mcp_get_code":{"code_sha256":"4700cde0d0e8a52a"}},{"arxiv_id":"2312.11598","paper":"/paper/skilldiffuser-interpretable-hierarchical","title":"SkillDiffuser: Interpretable Hierarchical Planning via Skill Abstractions in Diffusion-Based Task Execution","date":"2023-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Liang-ZX/SkillDiffuser","path":"skilldiffuser/hrl/eval_orig.py","file_url":"https://github.com/Liang-ZX/SkillDiffuser/blob/HEAD/skilldiffuser/hrl/eval_orig.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6bad17d71a8fc8bb","mcp_get_code":{"code_sha256":"6bad17d71a8fc8bb"}},{"arxiv_id":"2312.01097","paper":"/paper/planning-as-in-painting-a-diffusion-based","title":"Planning as In-Painting: A Diffusion-Based Embodied Task Planning Framework for Environments under Uncertainty","date":"2023-12-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"joeyy5588/planning-as-inpainting","path":"datasets/gridworld.py","file_url":"https://github.com/joeyy5588/planning-as-inpainting/blob/HEAD/datasets/gridworld.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6ceb7088925eb34f","mcp_get_code":{"code_sha256":"6ceb7088925eb34f"}},{"arxiv_id":"2311.13743","paper":"/paper/finme-a-performance-enhanced-large-language","title":"FinMem: A Performance-Enhanced LLM Trading Agent with Layered Memory and Character Design","date":"2023-11-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pipiku915/FinMem-LLM-StockTrading","path":"data-pipeline/07-metrics.py","file_url":"https://github.com/pipiku915/FinMem-LLM-StockTrading/blob/HEAD/data-pipeline/07-metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dae1a5ebf16c3cc2","mcp_get_code":{"code_sha256":"dae1a5ebf16c3cc2"}},{"arxiv_id":"2311.13743","paper":"/paper/finme-a-performance-enhanced-large-language","title":"FinMem: A Performance-Enhanced LLM Trading Agent with Layered Memory and Character Design","date":"2023-11-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pipiku915/FinMem-LLM-StockTrading","path":"data-pipeline/08-Wilcoxon-Test.py","file_url":"https://github.com/pipiku915/FinMem-LLM-StockTrading/blob/HEAD/data-pipeline/08-Wilcoxon-Test.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"db5525231f727fc6","mcp_get_code":{"code_sha256":"db5525231f727fc6"}},{"arxiv_id":"2310.12344","paper":"/paper/lacma-language-aligning-contrastive-learning","title":"LACMA: Language-Aligning Contrastive Learning with Meta-Actions for Embodied Instruction Following","date":"2023-10-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"joeyy5588/LACMA","path":"alfred/env/reward.py","file_url":"https://github.com/joeyy5588/LACMA/blob/HEAD/alfred/env/reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"12ba53fedf724ead","mcp_get_code":{"code_sha256":"12ba53fedf724ead"}},{"arxiv_id":"2310.00036","paper":"/paper/cleanba-a-reproducible-and-efficient","title":"Cleanba: A Reproducible and Efficient Distributed Reinforcement Learning Platform","date":"2023-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vwxyzjn/cleanba","path":"cleanba/legacy_scripts/cleanba_impala_envpool_impala_atari_wrapper.py","file_url":"https://github.com/vwxyzjn/cleanba/blob/HEAD/cleanba/legacy_scripts/cleanba_impala_envpool_impala_atari_wrapper.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"01a6c2cfba0e3978","mcp_get_code":{"code_sha256":"01a6c2cfba0e3978"}},{"arxiv_id":"2310.00036","paper":"/paper/cleanba-a-reproducible-and-efficient","title":"Cleanba: A Reproducible and Efficient Distributed Reinforcement Learning Platform","date":"2023-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vwxyzjn/cleanba","path":"cleanba/legacy_scripts/cleanba_impala_envpool_machado_atari_wrapper_last_action_reward.py","file_url":"https://github.com/vwxyzjn/cleanba/blob/HEAD/cleanba/legacy_scripts/cleanba_impala_envpool_machado_atari_wrapper_last_action_reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"a7cfe9ab0009183d","mcp_get_code":{"code_sha256":"a7cfe9ab0009183d"}},{"arxiv_id":"2309.17082","paper":"/paper/diffusion-models-as-stochastic-quantization","title":"Diffusion Models as Stochastic Quantization in Lattice Field Theory","date":"2023-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"anguswlx/dmassq","path":"mc.py","file_url":"https://github.com/anguswlx/dmassq/blob/HEAD/mc.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3e3c5437e484e594","mcp_get_code":{"code_sha256":"3e3c5437e484e594"}},{"arxiv_id":"2308.09387","paper":"/paper/multi-level-compositional-reasoning-for","title":"Multi-Level Compositional Reasoning for Interactive Instruction Following","date":"2023-08-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yonseivnl/mcr-agent","path":"env/reward.py","file_url":"https://github.com/yonseivnl/mcr-agent/blob/HEAD/env/reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"12ba53fedf724ead","mcp_get_code":{"code_sha256":"12ba53fedf724ead"}},{"arxiv_id":"2303.07551","paper":"/paper/merging-decision-transformers-weight","title":"Merging Decision Transformers: Weight Averaging for Forming Multi-Task Policies","date":"2023-03-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"daniellawson9999/merging-decision-transformers","path":"decision-transformer/dt_utils.py","file_url":"https://github.com/daniellawson9999/merging-decision-transformers/blob/HEAD/decision-transformer/dt_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ce4536d70d1e91b4","mcp_get_code":{"code_sha256":"ce4536d70d1e91b4"}},{"arxiv_id":"2302.12604","paper":"/paper/neural-laplace-control-for-continuous-time","title":"Neural Laplace Control for Continuous-time Delayed Systems","date":"2023-02-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vanderschaarlab/neurallaplacecontrol","path":"mppi_with_model.py","file_url":"https://github.com/vanderschaarlab/neurallaplacecontrol/blob/HEAD/mppi_with_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cdf86004b2feba7a","mcp_get_code":{"code_sha256":"cdf86004b2feba7a"}},{"arxiv_id":"2206.06289","paper":"/paper/silver-bullet-3d-at-maniskill-2021-learning","title":"Silver-Bullet-3D at ManiSkill 2021: Learning-from-Demonstrations and Heuristic Rule-based Methods for Object Manipulation","date":"2022-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"caiqi/silver-bullet-3d","path":"No_Restriction/user_solution_bucket.py","file_url":"https://github.com/caiqi/silver-bullet-3d/blob/HEAD/No_Restriction/user_solution_bucket.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0014de01d1e10092","mcp_get_code":{"code_sha256":"0014de01d1e10092"}},{"arxiv_id":"2109.05489","paper":"/paper/illuminating-diverse-neural-cellular-automata","title":"Illuminating Diverse Neural Cellular Automata for Level Generation","date":"2021-09-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"smearle/gym-pcgrl","path":"control_pcgrl/wrappers.py","file_url":"https://github.com/smearle/gym-pcgrl/blob/HEAD/control_pcgrl/wrappers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"493c58a0c3e0d196","mcp_get_code":{"code_sha256":"493c58a0c3e0d196"}},{"arxiv_id":"2107.04982","paper":"/paper/out-of-distribution-dynamics-detection-rl","title":"Out-of-Distribution Dynamics Detection: RL-Relevant Benchmarks and Results","date":"2021-07-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"modanesh/recurrent_implicit_quantile_networks","path":"autoregressive_control.py","file_url":"https://github.com/modanesh/recurrent_implicit_quantile_networks/blob/HEAD/autoregressive_control.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6b5f1c2ad4ca8a53","mcp_get_code":{"code_sha256":"6b5f1c2ad4ca8a53"}},{"arxiv_id":"2106.05087","paper":"/paper/who-is-the-strongest-enemy-towards-optimal","title":"Who Is the Strongest Enemy? Towards Optimal and Efficient Evasion Attacks in Deep RL","date":"2021-06-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"umd-huang-lab/paad_adv_rl","path":"code_atari/paad_rl/attack_robust/radial_train_attacker.py","file_url":"https://github.com/umd-huang-lab/paad_adv_rl/blob/HEAD/code_atari/paad_rl/attack_robust/radial_train_attacker.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2e86d0eab58b3a7b","mcp_get_code":{"code_sha256":"2e86d0eab58b3a7b"}},{"arxiv_id":"2106.03427","paper":"/paper/hierarchical-task-learning-from-language","title":"Hierarchical Task Learning from Language Instructions with Unified Transformers and Self-Monitoring","date":"2021-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"594zyc/HiTUT","path":"env/reward.py","file_url":"https://github.com/594zyc/HiTUT/blob/HEAD/env/reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"12ba53fedf724ead","mcp_get_code":{"code_sha256":"12ba53fedf724ead"}},{"arxiv_id":"2105.06453","paper":"/paper/episodic-transformer-for-vision-and-language","title":"Episodic Transformer for Vision-and-Language Navigation","date":"2021-05-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alexpashevich/E.T.","path":"alfred/env/reward.py","file_url":"https://github.com/alexpashevich/E.T./blob/HEAD/alfred/env/reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"12ba53fedf724ead","mcp_get_code":{"code_sha256":"12ba53fedf724ead"}},{"arxiv_id":"1912.01734","paper":"/paper/alfred-a-benchmark-for-interpreting-grounded","title":"ALFRED: A Benchmark for Interpreting Grounded Instructions for Everyday Tasks","date":"2019-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"askforalfred/alfred","path":"env/reward.py","file_url":"https://github.com/askforalfred/alfred/blob/HEAD/env/reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"12ba53fedf724ead","mcp_get_code":{"code_sha256":"12ba53fedf724ead"}},{"arxiv_id":"1909.13072","paper":"/paper/regression-planning-networks","title":"Regression Planning Networks","date":"2019-09-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"danfeiX/RPN","path":"rpn/grid_envs.py","file_url":"https://github.com/danfeiX/RPN/blob/HEAD/rpn/grid_envs.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"04faa2fa28055032","mcp_get_code":{"code_sha256":"04faa2fa28055032"}},{"arxiv_id":"1810.05546","paper":"/paper/uncertainty-in-neural-networks-bayesian","title":"Uncertainty in Neural Networks: Approximately Bayesian Ensembling","date":"2018-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"giarcieri/assessing-the-influence-of-models-on-the-performance-of-reinforcement-learning-algorithms","path":"play_one_step.py","file_url":"https://github.com/giarcieri/assessing-the-influence-of-models-on-the-performance-of-reinforcement-learning-algorithms/blob/HEAD/play_one_step.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a1a896e5eded3920","mcp_get_code":{"code_sha256":"a1a896e5eded3920"}},{"arxiv_id":"1810.00821","paper":"/paper/variational-discriminator-bottleneck","title":"Variational Discriminator Bottleneck: Improving Imitation Learning, Inverse RL, and GANs by Constraining Information Flow","date":"2018-10-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"reinforcement-learning-kr/lets-do-irl","path":"mujoco/gail/utils/utils.py","file_url":"https://github.com/reinforcement-learning-kr/lets-do-irl/blob/HEAD/mujoco/gail/utils/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b5a95d29758c57e3","mcp_get_code":{"code_sha256":"b5a95d29758c57e3"}},{"arxiv_id":"1802.10592","paper":"/paper/model-ensemble-trust-region-policy","title":"Model-Ensemble Trust-Region Policy Optimization","date":"2018-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thanard/me-trpo","path":"env_helpers.py","file_url":"https://github.com/thanard/me-trpo/blob/HEAD/env_helpers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1889cae48084f14c","mcp_get_code":{"code_sha256":"1889cae48084f14c"}}]}