{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/evaluate-episode","entry":"evaluate_episode","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":8,"n_papers_ran":4,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":6,"n_samples_ran":2,"n_samples_fingerprinted":0,"n_places":8,"n_places_pointer_only":0,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":1,"unverified":4},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2405.17098","paper":"/paper/q-value-regularized-transformer-for-offline","title":"Q-value Regularized Transformer for Offline Reinforcement Learning","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"charleshsc/qt","path":"decision_transformer/evaluation/evaluate_episodes.py","file_url":"https://github.com/charleshsc/qt/blob/HEAD/decision_transformer/evaluation/evaluate_episodes.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b0010b032f2fffe3","mcp_get_code":{"code_sha256":"b0010b032f2fffe3"}},{"arxiv_id":"2403.07309","paper":"/paper/reinforced-sequential-decision-making-for","title":"Reinforced Sequential Decision-Making for Sepsis Treatment: The POSNEGDM Framework with Mortality Classifier and Transformer","date":"2024-03-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dipeshtamboli/posnegdm-reinforced-sequential-decision-making-for-sepsis-treatment","path":"dualsight/evaluation/evaluate_episodes.py","file_url":"https://github.com/dipeshtamboli/posnegdm-reinforced-sequential-decision-making-for-sepsis-treatment/blob/HEAD/dualsight/evaluation/evaluate_episodes.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1ed718456cb9f7a0","mcp_get_code":{"code_sha256":"1ed718456cb9f7a0"}},{"arxiv_id":"2306.02224","paper":"/paper/auto-gpt-for-online-decision-making","title":"Auto-GPT for Online Decision Making: Benchmarks and Additional Opinions","date":"2023-06-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"younghuman/llmagent","path":"webshop/baseline_models/train_rl.py","file_url":"https://github.com/younghuman/llmagent/blob/HEAD/webshop/baseline_models/train_rl.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b65b7b0bbd7c3313","mcp_get_code":{"code_sha256":"b65b7b0bbd7c3313"}},{"arxiv_id":"2211.14655","paper":"/paper/how-crucial-is-transformer-in-decision","title":"How Crucial is Transformer in Decision Transformer?","date":"2022-11-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"max7born/decision-lstm","path":"src/decision_transformer/evaluation/evaluate_episodes.py","file_url":"https://github.com/max7born/decision-lstm/blob/HEAD/src/decision_transformer/evaluation/evaluate_episodes.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c3da8172ddacb573","mcp_get_code":{"code_sha256":"c3da8172ddacb573"}},{"arxiv_id":"2206.08353","paper":"/paper/towards-understanding-how-machines-can-learn","title":"Towards Understanding How Machines Can Learn Causal Overhypotheses","date":"2022-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cannylab/casual_overhypotheses","path":"models/decision-transformer/evaluation/evaluate_episodes.py","file_url":"https://github.com/cannylab/casual_overhypotheses/blob/HEAD/models/decision-transformer/evaluation/evaluate_episodes.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1ed718456cb9f7a0","mcp_get_code":{"code_sha256":"1ed718456cb9f7a0"}},{"arxiv_id":"2202.05607","paper":"/paper/online-decision-transformer","title":"Online Decision Transformer","date":"2022-02-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kzl/decision-transformer","path":"gym/decision_transformer/evaluation/evaluate_episodes.py","file_url":"https://github.com/kzl/decision-transformer/blob/HEAD/gym/decision_transformer/evaluation/evaluate_episodes.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"923bc87de491073f","mcp_get_code":{"code_sha256":"923bc87de491073f"}},{"arxiv_id":"2110.06206","paper":"/paper/starformer-transformer-with-state-action-1","title":"StARformer: Transformer with State-Action-Reward Representations for Visual Reinforcement Learning","date":"2021-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"elicassion/StARformer","path":"gym/evaluation/evaluate_episodes.py","file_url":"https://github.com/elicassion/StARformer/blob/HEAD/gym/evaluation/evaluate_episodes.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5d44a44afbb9c4ba","mcp_get_code":{"code_sha256":"5d44a44afbb9c4ba"}},{"arxiv_id":"2106.01345","paper":"/paper/decision-transformer-reinforcement-learning","title":"Decision Transformer: Reinforcement Learning via Sequence Modeling","date":"2021-06-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Amadeus979/decision-transformer","path":"gym/decision_transformer/evaluation/evaluate_episodes.py","file_url":"https://github.com/Amadeus979/decision-transformer/blob/HEAD/gym/decision_transformer/evaluation/evaluate_episodes.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1ed718456cb9f7a0","mcp_get_code":{"code_sha256":"1ed718456cb9f7a0"}}]}