{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/baseagent","entry":"BaseAgent","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":20,"n_papers_ran":6,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":21,"n_samples_ran":6,"n_samples_fingerprinted":0,"n_places":21,"n_places_pointer_only":6,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":6,"unverified":15},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.13522","paper":"/paper/arxiv-2608-13522","title":"Vero: Can AI Agents Build Formally Verified Software Repositories?","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"sunblaze-ucb/vero","path":"src/vero/generation/agents/base.py","file_url":"https://github.com/sunblaze-ucb/vero/blob/HEAD/src/vero/generation/agents/base.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5dbbe46a2b28ec5e","mcp_get_code":{"code_sha256":"5dbbe46a2b28ec5e"}},{"arxiv_id":"2608.09666","paper":"/paper/arxiv-2608-09666","title":"Open Evaluation Agent: Efficient and Promptable Evaluation of Visual Generative Models","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"Vchitect/Evaluation-Agent","path":"eval_agent/open_ended_eval.py","file_url":"https://github.com/Vchitect/Evaluation-Agent/blob/HEAD/eval_agent/open_ended_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"05cabef5f8ee8783","mcp_get_code":{"code_sha256":"05cabef5f8ee8783"}},{"arxiv_id":"2606.24428","paper":"/paper/arxiv-2606-24428","title":"Escaping the Self-Confirmation Trap: An Execute-Distill-Verify Paradigm for Agentic Experience Learning","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"shidingz/EDV","path":"src/edv/pipeline.py","file_url":"https://github.com/shidingz/EDV/blob/HEAD/src/edv/pipeline.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"348c63d8240bb48c","mcp_get_code":{"code_sha256":"348c63d8240bb48c"}},{"arxiv_id":"2605.27068","paper":"/paper/arxiv-2605-27068","title":"QUACK: Questioning, Understanding, and Auditing Communicated Knowledge in Multimodal Social Deduction Agents","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"AAAAA-Academia-Attractions/QUACK","path":"quack/agents/vlm_agent.py","file_url":"https://github.com/AAAAA-Academia-Attractions/QUACK/blob/HEAD/quack/agents/vlm_agent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ed456165ba9c9e7f","mcp_get_code":{"code_sha256":"ed456165ba9c9e7f"}},{"arxiv_id":"2605.14297","paper":"/paper/arxiv-2605-14297","title":"Policy Optimization in Hybrid Discrete-Continuous Action Spaces via Mixed Gradients","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"MatiasAlvo/hybrid-rl","path":"src/algorithms/hybrid/agents/hybrid_agent.py","file_url":"https://github.com/MatiasAlvo/hybrid-rl/blob/HEAD/src/algorithms/hybrid/agents/hybrid_agent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9df36bf4d58db933","mcp_get_code":{"code_sha256":"9df36bf4d58db933"}},{"arxiv_id":"2605.05863","paper":"/paper/arxiv-2605-05863","title":"SOPE: Stabilizing Off-Policy Evaluation for Online RL with Prior Data","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"CarloRomeo427/SOPE","path":"src/algos/agent_sope.py","file_url":"https://github.com/CarloRomeo427/SOPE/blob/HEAD/src/algos/agent_sope.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"26f0035bf4b9574f","mcp_get_code":{"code_sha256":"26f0035bf4b9574f"}},{"arxiv_id":"2604.15373","paper":"/paper/arxiv-2604-15373","title":"InfoChess: A Game of Adversarial Inference and a Laboratory for Quantifiable Information Control","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"murphyka/infochess","path":"agents/hiding_belief_vismax_agent.py","file_url":"https://github.com/murphyka/infochess/blob/HEAD/agents/hiding_belief_vismax_agent.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"72a0c90deaaae55c","mcp_get_code":{"code_sha256":"72a0c90deaaae55c"}},{"arxiv_id":"2601.19121","paper":"/paper/arxiv-2601-19121","title":"LLMs as Orchestrators: Constraint-Compliant Multi-Agent Optimization for Recommendation Systems","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"GuilinDev/Dual-Agents-Recommendation","path":"src/dualagent_rec.py","file_url":"https://github.com/GuilinDev/Dual-Agents-Recommendation/blob/HEAD/src/dualagent_rec.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"20d7db8280b5f1ce","mcp_get_code":{"code_sha256":"20d7db8280b5f1ce"}},{"arxiv_id":"2505.18943","paper":"/paper/metamind-modeling-human-social-thoughts-with","title":"MetaMind: Modeling Human Social Thoughts with Metacognitive Multi-Agent Systems","date":"2025-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xmzhangai/metamind","path":"agents/tom_agent.py","file_url":"https://github.com/xmzhangai/metamind/blob/HEAD/agents/tom_agent.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f00f2de3f0855743","mcp_get_code":{"code_sha256":"f00f2de3f0855743"}},{"arxiv_id":"2502.10937","paper":"/paper/scale-towards-collaborative-content-analysis","title":"SCALE: Towards Collaborative Content Analysis in Social Science with Large Language Model Agents and Human Intervention","date":"2025-02-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ChengshuaiZhao0/SCALE","path":"simulation/content_analysis_simulation.py","file_url":"https://github.com/ChengshuaiZhao0/SCALE/blob/HEAD/simulation/content_analysis_simulation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"49860cf76ad6194b","mcp_get_code":{"code_sha256":"49860cf76ad6194b"}},{"arxiv_id":"2412.03258","paper":"/paper/learning-on-one-mode-addressing-multi","title":"Learning on One Mode: Addressing Multi-Modality in Offline Reinforcement Learning","date":"2024-12-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MianchuWang/LOM","path":"agents/lom.py","file_url":"https://github.com/MianchuWang/LOM/blob/HEAD/agents/lom.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c9227a477207007c","mcp_get_code":{"code_sha256":"c9227a477207007c"}},{"arxiv_id":"2310.18803","paper":"/paper/weakly-coupled-deep-q-networks","title":"Weakly Coupled Deep Q-Networks","date":"2023-10-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ibrahim-elshar/WCDQN_NeurIPS","path":"src/Inv_control/WCDQN.py","file_url":"https://github.com/ibrahim-elshar/WCDQN_NeurIPS/blob/HEAD/src/Inv_control/WCDQN.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7f4eee504f02bf73","mcp_get_code":{"code_sha256":"7f4eee504f02bf73"}},{"arxiv_id":"2302.02662","paper":"/paper/grounding-large-language-models-in","title":"Grounding Large Language Models in Interactive Environments with Online Reinforcement Learning","date":"2023-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"flowersteam/grounding_llms_with_online_rl","path":"experiments/agents/ppo/llm_ppo_agent.py","file_url":"https://github.com/flowersteam/grounding_llms_with_online_rl/blob/HEAD/experiments/agents/ppo/llm_ppo_agent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3a32c056e4db9777","mcp_get_code":{"code_sha256":"3a32c056e4db9777"}},{"arxiv_id":"2211.10771","paper":"/paper/rl-boltzmann-generators-for-conformer","title":"RL Boltzmann Generators for Conformer Generation in Data-Sparse Environments","date":"2022-11-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yashpatel5400/clean_idp_rl","path":"main/agents/PPO_recurrent_agent.py","file_url":"https://github.com/yashpatel5400/clean_idp_rl/blob/HEAD/main/agents/PPO_recurrent_agent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ba3a139bcb1bfa01","mcp_get_code":{"code_sha256":"ba3a139bcb1bfa01"}},{"arxiv_id":"2205.12532","paper":"/paper/skill-machines-temporal-logic-composition-in","title":"Skill Machines: Temporal Logic Skill Composition in Reinforcement Learning","date":"2022-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"geraudnt/skill_machines","path":"sm.py","file_url":"https://github.com/geraudnt/skill_machines/blob/HEAD/sm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4bc65e0d8d7c8d26","mcp_get_code":{"code_sha256":"4bc65e0d8d7c8d26"}},{"arxiv_id":"2203.14936","paper":"/paper/fedvln-privacy-preserving-federated-vision","title":"FedVLN: Privacy-preserving Federated Vision-and-Language Navigation","date":"2022-03-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eric-ai-lab/FedVLN","path":"r2r_src/agent.py","file_url":"https://github.com/eric-ai-lab/FedVLN/blob/HEAD/r2r_src/agent.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a02da85c573a6d3a","mcp_get_code":{"code_sha256":"a02da85c573a6d3a"}},{"arxiv_id":"2007.14430","paper":"/paper/munchausen-reinforcement-learning","title":"Munchausen Reinforcement Learning","date":"2020-07-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lingweizhu/Pytorch-MunchausenActorCritic","path":"Munchausen_actorcritic_discrete/agent/munchausen_ac.py","file_url":"https://github.com/lingweizhu/Pytorch-MunchausenActorCritic/blob/HEAD/Munchausen_actorcritic_discrete/agent/munchausen_ac.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"636a9ac9309ad92c","mcp_get_code":{"code_sha256":"636a9ac9309ad92c"}},{"arxiv_id":"1707.06347","paper":"/paper/proximal-policy-optimization-algorithms","title":"Proximal Policy Optimization Algorithms","date":"2017-07-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dmiu-shell/deeprl-shell","path":"deep_rl/agent/PPO_agent.py","file_url":"https://github.com/dmiu-shell/deeprl-shell/blob/HEAD/deep_rl/agent/PPO_agent.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"38803d98e366da6d","mcp_get_code":{"code_sha256":"38803d98e366da6d"}},{"arxiv_id":"1707.06347","paper":"/paper/proximal-policy-optimization-algorithms","title":"Proximal Policy Optimization Algorithms","date":"2017-07-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hamishs/JAX-RL","path":"src/jax_rl/algorithms/ppo.py","file_url":"https://github.com/hamishs/JAX-RL/blob/HEAD/src/jax_rl/algorithms/ppo.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2dd687e050a61476","mcp_get_code":{"code_sha256":"2dd687e050a61476"}},{"arxiv_id":"1509.02971","paper":"/paper/continuous-control-with-deep-reinforcement","title":"Continuous control with deep reinforcement learning","date":"2015-09-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"majercakdavid/gym-virtual-quant-trading","path":"agents/DDPG.py","file_url":"https://github.com/majercakdavid/gym-virtual-quant-trading/blob/HEAD/agents/DDPG.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"ddcc51cfeb97e2a4","mcp_get_code":{"code_sha256":"ddcc51cfeb97e2a4"}},{"arxiv_id":"1003.0146","paper":"/paper/a-contextual-bandit-approach-to-personalized","title":"A Contextual-Bandit Approach to Personalized News Article Recommendation","date":"2010-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"danilprov/batch-bandits","path":"CMAB/LinUCB.py","file_url":"https://github.com/danilprov/batch-bandits/blob/HEAD/CMAB/LinUCB.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2df4798ead73ae05","mcp_get_code":{"code_sha256":"2df4798ead73ae05"}}]}