{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/imitation-learning/papers/5","list_of":"/task/imitation-learning","task":"Imitation Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":22,"rows_per_page":100,"rows":[401,500],"of":2122,"counts":{"archive_papers_tagged":2122,"with_a_code_link":691,"where_syntology_ran_a_sample":234,"not_listed_spam_title":0,"listed":2122,"listed_where_code_ran":234,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":191,"every_run_a_failure_of_syntologys_instrument":43,"listed_with_a_run_with_no_instrument_failure":191,"listed_every_run_a_failure_of_syntologys_instrument":43,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/imitation-learning","prev":"/task/imitation-learning/papers/4","next":"/task/imitation-learning/papers/6","papers":[{"url":"/paper/inferring-versatile-behavior-from","slug":"inferring-versatile-behavior-from","title":"Inferring Versatile Behavior from Demonstrations by Matching Geometric Descriptors","date":"2022-10-17","arxiv_id":"2210.08121","repositories_listed":1,"syntology":null},{"url":"/paper/frame-mining-a-free-lunch-for-learning","slug":"frame-mining-a-free-lunch-for-learning","title":"Frame Mining: a Free Lunch for Learning Robotic Manipulation from 3D Point Clouds","date":"2022-10-14","arxiv_id":"2210.07442","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/frame-mining-a-free-lunch-for-learning#ran","syntology_url":"https://syntology.ai/paper/2210.07442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07442"}},"official":{"repos":["xuanlinli17/corl_22_frame_mining"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-imitation-learning-for-urban","slug":"model-based-imitation-learning-for-urban","title":"Model-Based Imitation Learning for Urban Driving","date":"2022-10-14","arxiv_id":"2210.07729","repositories_listed":1,"syntology":{"n":23,"n_ran":21,"n_constructed":13,"n_ran_checked":13,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"21 ran (of which 13 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/model-based-imitation-learning-for-urban#ran","syntology_url":"https://syntology.ai/paper/2210.07729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07729"}},"official":{"repos":["wayveai/mile"],"state":"official (archive's flag): 21 ran","n_ran":21,"n_constructed":13,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/travel-the-same-path-a-novel-tsp-solving","slug":"travel-the-same-path-a-novel-tsp-solving","title":"Travel the Same Path: A Novel TSP Solving Strategy","date":"2022-10-12","arxiv_id":"2210.05906","repositories_listed":1,"syntology":null},{"url":"/paper/markup-to-image-diffusion-models-with","slug":"markup-to-image-diffusion-models-with","title":"Markup-to-Image Diffusion Models with Scheduled Sampling","date":"2022-10-11","arxiv_id":"2210.05147","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/markup-to-image-diffusion-models-with#ran","syntology_url":"https://syntology.ai/paper/2210.05147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05147"}},"official":{"repos":["da03/markup2im"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/don-t-copy-the-teacher-data-and-model","slug":"don-t-copy-the-teacher-data-and-model","title":"Don't Copy the Teacher: Data and Model Challenges in Embodied Dialogue","date":"2022-10-10","arxiv_id":"2210.04443","repositories_listed":1,"syntology":null},{"url":"/paper/cw-erm-improving-autonomous-driving-planning","slug":"cw-erm-improving-autonomous-driving-planning","title":"CW-ERM: Improving Autonomous Driving Planning with Closed-loop Weighted Empirical Risk Minimization","date":"2022-10-05","arxiv_id":"2210.02174","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-adversarial-inverse","slug":"hierarchical-adversarial-inverse","title":"Option-Aware Adversarial Inverse Reinforcement Learning for Robotic Control","date":"2022-10-05","arxiv_id":"2210.01969","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-reward-shaping-for-a-robotic","slug":"unsupervised-reward-shaping-for-a-robotic","title":"Unsupervised Reward Shaping for a Robotic Sequential Picking Task from Visual Observations in a Logistics Scenario","date":"2022-09-25","arxiv_id":"2209.12350","repositories_listed":1,"syntology":null},{"url":"/paper/latent-plans-for-task-agnostic-offline","slug":"latent-plans-for-task-agnostic-offline","title":"Latent Plans for Task-Agnostic Offline Reinforcement Learning","date":"2022-09-19","arxiv_id":"2209.08959","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-walk-by-steering-perceptive","slug":"learning-to-walk-by-steering-perceptive","title":"Learning to Walk by Steering: Perceptive Quadrupedal Locomotion in Dynamic Environments","date":"2022-09-19","arxiv_id":"2209.09233","repositories_listed":1,"syntology":null},{"url":"/paper/imitrob-imitation-learning-dataset-for","slug":"imitrob-imitation-learning-dataset-for","title":"Imitrob: Imitation Learning Dataset for Training and Evaluating 6D Object Pose Estimators","date":"2022-09-16","arxiv_id":"2209.07976","repositories_listed":1,"syntology":null},{"url":"/paper/solving-the-baby-intuitions-benchmark-with-a","slug":"solving-the-baby-intuitions-benchmark-with-a","title":"Solving the Baby Intuitions Benchmark with a Hierarchically Bayesian Theory of Mind","date":"2022-08-04","arxiv_id":"2208.02914","repositories_listed":1,"syntology":null},{"url":"/paper/sequence-model-imitation-learning-with","slug":"sequence-model-imitation-learning-with","title":"Sequence Model Imitation Learning with Unobserved Contexts","date":"2022-08-03","arxiv_id":"2208.02225","repositories_listed":1,"syntology":null},{"url":"/paper/improved-policy-optimization-for-online","slug":"improved-policy-optimization-for-online","title":"Improved Policy Optimization for Online Imitation Learning","date":"2022-07-29","arxiv_id":"2208.00088","repositories_listed":1,"syntology":null},{"url":"/paper/learning-soccer-juggling-skills-with-layer","slug":"learning-soccer-juggling-skills-with-layer","title":"Learning Soccer Juggling Skills with Layer-wise Mixture-of-Experts","date":"2022-07-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-accelerate-approximate-methods","slug":"learning-to-accelerate-approximate-methods","title":"Learning to Accelerate Approximate Methods for Solving Integer Programming via Early Fixing","date":"2022-07-05","arxiv_id":"2207.02087","repositories_listed":1,"syntology":null},{"url":"/paper/target-absent-human-attention","slug":"target-absent-human-attention","title":"Target-absent Human Attention","date":"2022-07-04","arxiv_id":"2207.01166","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/target-absent-human-attention#ran","syntology_url":"https://syntology.ai/paper/2207.01166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01166"}},"official":{"repos":["cvlab-stonybrook/target-absent-human-attention"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-policy-optimization-with-generalist","slug":"improving-policy-optimization-with-generalist","title":"Improving Policy Optimization with Generalist-Specialist Learning","date":"2022-06-26","arxiv_id":"2206.12984","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-for-generalizable-self","slug":"imitation-learning-for-generalizable-self","title":"Imitation Learning for Generalizable Self-driving Policy with Sim-to-real Transfer","date":"2022-06-22","arxiv_id":"2206.10797","repositories_listed":1,"syntology":null},{"url":"/paper/nocturne-a-scalable-driving-benchmark-for","slug":"nocturne-a-scalable-driving-benchmark-for","title":"Nocturne: a scalable driving benchmark for bringing multi-agent learning one step closer to the real world","date":"2022-06-20","arxiv_id":"2206.09889","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nocturne-a-scalable-driving-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2206.09889","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.09889"}},"official":{"repos":["facebookresearch/nocturne"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-imitation-learning-against-variations","slug":"robust-imitation-learning-against-variations","title":"Robust Imitation Learning against Variations in Environment Dynamics","date":"2022-06-19","arxiv_id":"2206.09314","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-imitation-learning-against-variations#ran","syntology_url":"https://syntology.ai/paper/2206.09314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.09314"}},"official":{"repos":["jongseongchae/rime"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-inverse-reinforcement-learning-for-route","slug":"deep-inverse-reinforcement-learning-for-route","title":"A deep inverse reinforcement learning approach to route choice modeling with context-dependent rewards","date":"2022-06-18","arxiv_id":"2206.10598","repositories_listed":1,"syntology":null},{"url":"/paper/silver-bullet-3d-at-maniskill-2021-learning","slug":"silver-bullet-3d-at-maniskill-2021-learning","title":"Silver-Bullet-3D at ManiSkill 2021: Learning-from-Demonstrations and Heuristic Rule-based Methods for Object Manipulation","date":"2022-06-13","arxiv_id":"2206.06289","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":9,"n_ran_checked":10,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":12,"phrase":"13 ran (of which 9 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/silver-bullet-3d-at-maniskill-2021-learning#ran","syntology_url":"https://syntology.ai/paper/2206.06289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.06289"}},"official":{"repos":["caiqi/silver-bullet-3d"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":9,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/case-based-inverse-reinforcement-learning","slug":"case-based-inverse-reinforcement-learning","title":"Case-Based Inverse Reinforcement Learning Using Temporal Coherence","date":"2022-06-12","arxiv_id":"2206.05827","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-via-differentiable-physics","slug":"imitation-learning-via-differentiable-physics","title":"Imitation Learning via Differentiable Physics","date":"2022-06-10","arxiv_id":"2206.04873","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imitation-learning-via-differentiable-physics#ran","syntology_url":"https://syntology.ai/paper/2206.04873","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.04873"}},"official":{"repos":["sail-sg/ild"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/minimax-optimal-online-imitation-learning-via","slug":"minimax-optimal-online-imitation-learning-via","title":"Minimax Optimal Online Imitation Learning via Replay Estimation","date":"2022-05-30","arxiv_id":"2205.15397","repositories_listed":1,"syntology":null},{"url":"/paper/tasil-taylor-series-imitation-learning","slug":"tasil-taylor-series-imitation-learning","title":"TaSIL: Taylor Series Imitation Learning","date":"2022-05-30","arxiv_id":"2205.14812","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tasil-taylor-series-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2205.14812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14812"}},"official":{"repos":["unstable-zeros/tasil"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-branch-and-bound","slug":"reinforcement-learning-for-branch-and-bound","title":"Reinforcement Learning for Branch-and-Bound Optimisation using Retrospective Trajectories","date":"2022-05-28","arxiv_id":"2205.14345","repositories_listed":1,"syntology":null},{"url":"/paper/chain-of-thought-imitation-with-procedure","slug":"chain-of-thought-imitation-with-procedure","title":"Chain of Thought Imitation with Procedure Cloning","date":"2022-05-22","arxiv_id":"2205.10816","repositories_listed":1,"syntology":null},{"url":"/paper/ase-large-scale-reusable-adversarial-skill","slug":"ase-large-scale-reusable-adversarial-skill","title":"ASE: Large-Scale Reusable Adversarial Skill Embeddings for Physically Simulated Characters","date":"2022-05-04","arxiv_id":"2205.01906","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ase-large-scale-reusable-adversarial-skill#ran","syntology_url":"https://syntology.ai/paper/2205.01906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01906"}},"official":null}},{"url":"/paper/king-generating-safety-critical-driving","slug":"king-generating-safety-critical-driving","title":"KING: Generating Safety-Critical Driving Scenarios for Robust Imitation via Kinematics Gradients","date":"2022-04-28","arxiv_id":"2204.13683","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/king-generating-safety-critical-driving#ran","syntology_url":"https://syntology.ai/paper/2204.13683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.13683"}},"official":{"repos":["autonomousvision/king"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-from-observations-under-1","slug":"imitation-learning-from-observations-under-1","title":"Imitation Learning from Observations under Transition Model Disparity","date":"2022-04-25","arxiv_id":"2204.11446","repositories_listed":1,"syntology":null},{"url":"/paper/the-boltzmann-policy-distribution-accounting-1","slug":"the-boltzmann-policy-distribution-accounting-1","title":"The Boltzmann Policy Distribution: Accounting for Systematic Suboptimality in Human Models","date":"2022-04-22","arxiv_id":"2204.10759","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-boltzmann-policy-distribution-accounting-1#ran","syntology_url":"https://syntology.ai/paper/2204.10759","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.10759"}},"official":{"repos":["cassidylaidlaw/boltzmann-policy-distribution"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/non-parallel-text-style-transfer-with-self-1","slug":"non-parallel-text-style-transfer-with-self-1","title":"Non-Parallel Text Style Transfer with Self-Parallel Supervision","date":"2022-04-18","arxiv_id":"2204.08123","repositories_listed":1,"syntology":null},{"url":"/paper/divide-conquer-imitation-learning","slug":"divide-conquer-imitation-learning","title":"Divide & Conquer Imitation Learning","date":"2022-04-15","arxiv_id":"2204.07404","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-effectiveness-of-corrective","slug":"evaluating-the-effectiveness-of-corrective","title":"Evaluating the Effectiveness of Corrective Demonstrations and a Low-Cost Sensor for Dexterous Manipulation","date":"2022-04-15","arxiv_id":"2204.07631","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-game-playing-agents-with","slug":"understanding-game-playing-agents-with","title":"Understanding Game-Playing Agents with Natural Language Annotations","date":"2022-04-15","arxiv_id":"2204.07531","repositories_listed":1,"syntology":null},{"url":"/paper/action-conditioned-contrastive-policy","slug":"action-conditioned-contrastive-policy","title":"Learning to Drive by Watching YouTube Videos: Action-Conditioned Contrastive Policy Pretraining","date":"2022-04-05","arxiv_id":"2204.02393","repositories_listed":1,"syntology":null},{"url":"/paper/gail-pt-a-generic-intelligent-penetration","slug":"gail-pt-a-generic-intelligent-penetration","title":"GAIL-PT: A Generic Intelligent Penetration Testing Framework with Generative Adversarial Imitation Learning","date":"2022-04-05","arxiv_id":"2204.01975","repositories_listed":1,"syntology":null},{"url":"/paper/why-exposure-bias-matters-an-imitation","slug":"why-exposure-bias-matters-an-imitation","title":"Why Exposure Bias Matters: An Imitation Learning Perspective of Error Accumulation in Language Generation","date":"2022-04-03","arxiv_id":"2204.01171","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/why-exposure-bias-matters-an-imitation#ran","syntology_url":"https://syntology.ai/paper/2204.01171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.01171"}},"official":{"repos":["kushalarora/quantifying_exposure_bias"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/modular-adaptive-policy-selection-for-multi","slug":"modular-adaptive-policy-selection-for-multi","title":"Modular Adaptive Policy Selection for Multi-Task Imitation Learning through Task Division","date":"2022-03-28","arxiv_id":"2203.14855","repositories_listed":1,"syntology":null},{"url":"/paper/a-visual-navigation-perspective-for-category","slug":"a-visual-navigation-perspective-for-category","title":"A Visual Navigation Perspective for Category-Level Object Pose Estimation","date":"2022-03-25","arxiv_id":"2203.13572","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 2 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-visual-navigation-perspective-for-category#ran","syntology_url":"https://syntology.ai/paper/2203.13572","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13572"}},"official":{"repos":["wrld/visual_navigation_pose_estimation"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/teachable-reinforcement-learning-via-advice-1","slug":"teachable-reinforcement-learning-via-advice-1","title":"Teachable Reinforcement Learning via Advice Distillation","date":"2022-03-19","arxiv_id":"2203.11197","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/teachable-reinforcement-learning-via-advice-1#ran","syntology_url":"https://syntology.ai/paper/2203.11197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.11197"}},"official":{"repos":["rll-research/teachable"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/motionaug-augmentation-with-physical","slug":"motionaug-augmentation-with-physical","title":"MotionAug: Augmentation with Physical Correction for Human Motion Prediction","date":"2022-03-17","arxiv_id":"2203.09116","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/motionaug-augmentation-with-physical#ran","syntology_url":"https://syntology.ai/paper/2203.09116","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09116"}},"official":{"repos":["meaten/motionaug"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-humans-combining-imitation-and","slug":"learning-from-humans-combining-imitation-and","title":"Combining imitation and deep reinforcement learning to accomplish human-level performance on a virtual foraging task","date":"2022-03-11","arxiv_id":"2203.06250","repositories_listed":1,"syntology":null},{"url":"/paper/mirror-differentiable-deep-social-projection","slug":"mirror-differentiable-deep-social-projection","title":"MIRROR: Differentiable Deep Social Projection for Assistive Human-Robot Communication","date":"2022-03-06","arxiv_id":"2203.02877","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-gap-between-learning-in-discrete","slug":"bridging-the-gap-between-learning-in-discrete","title":"Bridging the Gap Between Learning in Discrete and Continuous Environments for Vision-and-Language Navigation","date":"2022-03-05","arxiv_id":"2203.02764","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-the-gap-between-learning-in-discrete#ran","syntology_url":"https://syntology.ai/paper/2203.02764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.02764"}},"official":{"repos":["yiconghong/discrete-continuous-vln"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fail-safe-generative-adversarial-imitation","slug":"fail-safe-generative-adversarial-imitation","title":"Fail-Safe Adversarial Generative Imitation Learning","date":"2022-03-03","arxiv_id":"2203.01696","repositories_listed":1,"syntology":null},{"url":"/paper/lisa-learning-interpretable-skill","slug":"lisa-learning-interpretable-skill","title":"LISA: Learning Interpretable Skill Abstractions from Language","date":"2022-02-28","arxiv_id":"2203.00054","repositories_listed":1,"syntology":null},{"url":"/paper/all-you-need-is-supervised-learning-from","slug":"all-you-need-is-supervised-learning-from","title":"All You Need Is Supervised Learning: From Imitation Learning to Meta-RL With Upside Down RL","date":"2022-02-24","arxiv_id":"2202.11960","repositories_listed":1,"syntology":null},{"url":"/paper/robust-learning-from-observation-with-model","slug":"robust-learning-from-observation-with-model","title":"Robust Learning from Observation with Model Misspecification","date":"2022-02-12","arxiv_id":"2202.06003","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-fusion-for-sensorimotor","slug":"multi-modal-fusion-for-sensorimotor","title":"Multi-Modal Fusion for Sensorimotor Coordination in Steering Angle Prediction","date":"2022-02-11","arxiv_id":"2202.05500","repositories_listed":1,"syntology":null},{"url":"/paper/revolver-continuous-evolutionary-models-for","slug":"revolver-continuous-evolutionary-models-for","title":"REvolveR: Continuous Evolutionary Models for Robot-to-robot Policy Transfer","date":"2022-02-10","arxiv_id":"2202.05244","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revolver-continuous-evolutionary-models-for#ran","syntology_url":"https://syntology.ai/paper/2202.05244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.05244"}},"official":{"repos":["xingyul/revolver"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bayesian-nonparametrics-for-offline-skill","slug":"bayesian-nonparametrics-for-offline-skill","title":"Bayesian Nonparametrics for Offline Skill Discovery","date":"2022-02-09","arxiv_id":"2202.04675","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bayesian-nonparametrics-for-offline-skill#ran","syntology_url":"https://syntology.ai/paper/2202.04675","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04675"}},"official":{"repos":["layer6ai-labs/bnpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-by-state-only-distribution","slug":"imitation-learning-by-state-only-distribution","title":"Imitation Learning by State-Only Distribution Matching","date":"2022-02-09","arxiv_id":"2202.04332","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imitation-learning-by-state-only-distribution#ran","syntology_url":"https://syntology.ai/paper/2202.04332","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04332"}},"official":{"repos":["FeMa42/soil-tdm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-trained-language-models-for-interactive","slug":"pre-trained-language-models-for-interactive","title":"Pre-Trained Language Models for Interactive Decision-Making","date":"2022-02-03","arxiv_id":"2202.01771","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/pre-trained-language-models-for-interactive#ran","syntology_url":"https://syntology.ai/paper/2202.01771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.01771"}},"official":null}},{"url":"/paper/causal-imitation-learning-under-temporally","slug":"causal-imitation-learning-under-temporally","title":"Causal Imitation Learning under Temporally Correlated Noise","date":"2022-02-02","arxiv_id":"2202.01312","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-by-estimating-expertise-of","slug":"imitation-learning-by-estimating-expertise-of","title":"Imitation Learning by Estimating Expertise of Demonstrators","date":"2022-02-02","arxiv_id":"2202.01288","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-learning-by-estimating-expertise-of#ran","syntology_url":"https://syntology.ai/paper/2202.01288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.01288"}},"official":{"repos":["stanford-iliad/ileed"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-general-evolution-inspired-reward-function","slug":"a-general-evolution-inspired-reward-function","title":"A General, Evolution-Inspired Reward Function for Social Robotics","date":"2022-02-01","arxiv_id":"2202.00617","repositories_listed":1,"syntology":null},{"url":"/paper/burst-dependent-plasticity-and-dendritic","slug":"burst-dependent-plasticity-and-dendritic","title":"Burst-dependent plasticity and dendritic amplification support target-based learning and hierarchical imitation learning","date":"2022-01-27","arxiv_id":"2201.11717","repositories_listed":1,"syntology":null},{"url":"/paper/deecap-dynamic-early-exiting-for-efficient","slug":"deecap-dynamic-early-exiting-for-efficient","title":"DeeCap: Dynamic Early Exiting for Efficient Image Captioning","date":"2022-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-guided-play-a-scheduled","slug":"learning-from-guided-play-a-scheduled","title":"Learning from Guided Play: A Scheduled Hierarchical Approach for Improving Exploration in Adversarial Imitation Learning","date":"2021-12-16","arxiv_id":"2112.08932","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-from-guided-play-a-scheduled#ran","syntology_url":"https://syntology.ai/paper/2112.08932","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.08932"}},"official":{"repos":["utiasstars/lfgp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-guide-and-to-be-guided-in-the-1","slug":"learning-to-guide-and-to-be-guided-in-the-1","title":"Learning to Guide and to Be Guided in the Architect-Builder Problem","date":"2021-12-14","arxiv_id":"2112.07342","repositories_listed":1,"syntology":null},{"url":"/paper/deterministic-and-discriminative-imitation-d2","slug":"deterministic-and-discriminative-imitation-d2","title":"Deterministic and Discriminative Imitation (D2-Imitation): Revisiting Adversarial Imitation for Sample Efficiency","date":"2021-12-11","arxiv_id":"2112.06054","repositories_listed":1,"syntology":null},{"url":"/paper/causal-imitative-model-for-autonomous-driving","slug":"causal-imitative-model-for-autonomous-driving","title":"Causal Imitative Model for Autonomous Driving","date":"2021-12-07","arxiv_id":"2112.03908","repositories_listed":1,"syntology":null},{"url":"/paper/combining-learning-from-human-feedback-and","slug":"combining-learning-from-human-feedback-and","title":"Combining Learning from Human Feedback and Knowledge Engineering to Solve Hierarchical Tasks in Minecraft","date":"2021-12-07","arxiv_id":"2112.03482","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/combining-learning-from-human-feedback-and#ran","syntology_url":"https://syntology.ai/paper/2112.03482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.03482"}},"official":{"repos":["viniciusguigo/kairos_minerl_basalt"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/calvin-a-benchmark-for-language-conditioned","slug":"calvin-a-benchmark-for-language-conditioned","title":"CALVIN: A Benchmark for Language-Conditioned Policy Learning for Long-Horizon Robot Manipulation Tasks","date":"2021-12-06","arxiv_id":"2112.03227","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/calvin-a-benchmark-for-language-conditioned#ran","syntology_url":"https://syntology.ai/paper/2112.03227","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.03227"}},"official":{"repos":["mees/calvin"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/the-surprising-effectiveness-of","slug":"the-surprising-effectiveness-of","title":"The Surprising Effectiveness of Representation Learning for Visual Imitation","date":"2021-12-02","arxiv_id":"2112.01511","repositories_listed":1,"syntology":null},{"url":"/paper/a-general-language-assistant-as-a-laboratory","slug":"a-general-language-assistant-as-a-laboratory","title":"A General Language Assistant as a Laboratory for Alignment","date":"2021-12-01","arxiv_id":"2112.00861","repositories_listed":1,"syntology":null},{"url":"/paper/solving-graph-based-public-goods-games-with","slug":"solving-graph-based-public-goods-games-with","title":"Solving Graph-based Public Goods Games with Tree Search and Imitation Learning","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/generalized-decision-transformer-for-offline","slug":"generalized-decision-transformer-for-offline","title":"Generalized Decision Transformer for Offline Hindsight Information Matching","date":"2021-11-19","arxiv_id":"2111.10364","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalized-decision-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2111.10364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.10364"}},"official":{"repos":["frt03/generalized_dt"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/distilling-motion-planner-augmented-policies","slug":"distilling-motion-planner-augmented-policies","title":"Distilling Motion Planner Augmented Policies into Visual Control Policies for Robot Manipulation","date":"2021-11-11","arxiv_id":"2111.06383","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/distilling-motion-planner-augmented-policies#ran","syntology_url":"https://syntology.ai/paper/2111.06383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.06383"}},"official":{"repos":["clvrai/mopa-pd"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/lila-language-informed-latent-actions","slug":"lila-language-informed-latent-actions","title":"LILA: Language-Informed Latent Actions","date":"2021-11-05","arxiv_id":"2111.03205","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lila-language-informed-latent-actions#ran","syntology_url":"https://syntology.ai/paper/2111.03205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.03205"}},"official":{"repos":["siddk/lila"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/rlds-an-ecosystem-to-generate-share-and-use","slug":"rlds-an-ecosystem-to-generate-share-and-use","title":"RLDS: an Ecosystem to Generate, Share and Use Datasets in Reinforcement Learning","date":"2021-11-04","arxiv_id":"2111.02767","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rlds-an-ecosystem-to-generate-share-and-use#ran","syntology_url":"https://syntology.ai/paper/2111.02767","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.02767"}},"official":{"repos":["google-research/rlds"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/curriculum-offline-imitation-learning","slug":"curriculum-offline-imitation-learning","title":"Curriculum Offline Imitation Learning","date":"2021-11-03","arxiv_id":"2111.02056","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/curriculum-offline-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2111.02056","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.02056"}},"official":{"repos":["apexrl/coil"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/object-aware-regularization-for-addressing","slug":"object-aware-regularization-for-addressing","title":"Object-Aware Regularization for Addressing Causal Confusion in Imitation Learning","date":"2021-10-27","arxiv_id":"2110.14118","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/object-aware-regularization-for-addressing#ran","syntology_url":"https://syntology.ai/paper/2110.14118","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14118"}},"official":{"repos":["alinlab/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/trail-near-optimal-imitation-learning-with-1","slug":"trail-near-optimal-imitation-learning-with-1","title":"TRAIL: Near-Optimal Imitation Learning with Suboptimal Data","date":"2021-10-27","arxiv_id":"2110.14770","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-decision-making-for-active-object","slug":"sequential-decision-making-for-active-object","title":"Sequential Voting with Relational Box Fields for Active Object Detection","date":"2021-10-21","arxiv_id":"2110.11524","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-control-with-action-quantization-1","slug":"continuous-control-with-action-quantization-1","title":"Continuous Control with Action Quantization from Demonstrations","date":"2021-10-19","arxiv_id":"2110.10149","repositories_listed":1,"syntology":null},{"url":"/paper/film-following-instructions-in-language-with-1","slug":"film-following-instructions-in-language-with-1","title":"FILM: Following Instructions in Language with Modular Methods","date":"2021-10-12","arxiv_id":"2110.07342","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":1,"n_ran_checked":2,"n_instrument":7,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":10,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 7 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/film-following-instructions-in-language-with-1#ran","syntology_url":"https://syntology.ai/paper/2110.07342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.07342"}},"official":{"repos":["soyeonm/film"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/starformer-transformer-with-state-action-1","slug":"starformer-transformer-with-state-action-1","title":"StARformer: Transformer with State-Action-Reward Representations for Visual Reinforcement Learning","date":"2021-10-12","arxiv_id":"2110.06206","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/starformer-transformer-with-state-action-1#ran","syntology_url":"https://syntology.ai/paper/2110.06206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.06206"}},"official":{"repos":["elicassion/StARformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/autonomous-racing-using-a-hybrid-imitation","slug":"autonomous-racing-using-a-hybrid-imitation","title":"Autonomous Racing using a Hybrid Imitation-Reinforcement Learning Architecture","date":"2021-10-11","arxiv_id":"2110.05437","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-in-two-player-zero-sum","slug":"reinforcement-learning-in-two-player-zero-sum","title":"Reinforcement Learning In Two Player Zero Sum Simultaneous Action Games","date":"2021-10-10","arxiv_id":"2110.04835","repositories_listed":1,"syntology":null},{"url":"/paper/cross-domain-imitation-learning-via-optimal","slug":"cross-domain-imitation-learning-via-optimal","title":"Cross-Domain Imitation Learning via Optimal Transport","date":"2021-10-07","arxiv_id":"2110.03684","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-discovery-and-description-of-human","slug":"automatic-discovery-and-description-of-human","title":"Automatic Discovery and Description of Human Planning Strategies","date":"2021-09-29","arxiv_id":"2109.14493","repositories_listed":1,"syntology":null},{"url":"/paper/emergent-communication-at-scale","slug":"emergent-communication-at-scale","title":"Emergent Communication at Scale","date":"2021-09-29","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cliport-what-and-where-pathways-for-robotic","slug":"cliport-what-and-where-pathways-for-robotic","title":"CLIPort: What and Where Pathways for Robotic Manipulation","date":"2021-09-24","arxiv_id":"2109.12098","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-of-stabilizing-policies","slug":"imitation-learning-of-stabilizing-policies","title":"Imitation Learning of Stabilizing Policies for Nonlinear Systems","date":"2021-09-22","arxiv_id":"2109.10854","repositories_listed":1,"syntology":null},{"url":"/paper/reactive-and-safe-road-user-simulations-using","slug":"reactive-and-safe-road-user-simulations-using","title":"Reactive and Safe Road User Simulations using Neural Barrier Certificates","date":"2021-09-14","arxiv_id":"2109.06689","repositories_listed":1,"syntology":null},{"url":"/paper/cross-domain-robot-imitation-with-invariant","slug":"cross-domain-robot-imitation-with-invariant","title":"Cross Domain Robot Imitation with Invariant Representation","date":"2021-09-13","arxiv_id":"2109.05940","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cross-domain-robot-imitation-with-invariant#ran","syntology_url":"https://syntology.ai/paper/2109.05940","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.05940"}},"official":{"repos":["zhaohengyin/irgail_example"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-selective-communication-for-multi","slug":"learning-selective-communication-for-multi","title":"Learning Selective Communication for Multi-Agent Path Finding","date":"2021-09-12","arxiv_id":"2109.05413","repositories_listed":1,"syntology":null},{"url":"/paper/generating-self-contained-and-summary-centric","slug":"generating-self-contained-and-summary-centric","title":"Generating Self-Contained and Summary-Centric Question Answer Pairs via Differentiable Reward Imitation Learning","date":"2021-09-10","arxiv_id":"2109.04689","repositories_listed":1,"syntology":null},{"url":"/paper/neat-neural-attention-fields-for-end-to-end","slug":"neat-neural-attention-fields-for-end-to-end","title":"NEAT: Neural Attention Fields for End-to-End Autonomous Driving","date":"2021-09-09","arxiv_id":"2109.04456","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/neat-neural-attention-fields-for-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2109.04456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.04456"}},"official":{"repos":["autonomousvision/neat"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/imital-learning-active-learning-strategies","slug":"imital-learning-active-learning-strategies","title":"ImitAL: Learning Active Learning Strategies from Synthetic Data","date":"2021-08-17","arxiv_id":"2108.07670","repositories_listed":1,"syntology":null},{"url":"/paper/dexmv-imitation-learning-for-dexterous","slug":"dexmv-imitation-learning-for-dexterous","title":"DexMV: Imitation Learning for Dexterous Manipulation from Human Videos","date":"2021-08-12","arxiv_id":"2108.05877","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/dexmv-imitation-learning-for-dexterous#ran","syntology_url":"https://syntology.ai/paper/2108.05877","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.05877"}},"official":{"repos":["yzqin/dexmv-sim"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-by-reinforcement-learning","slug":"imitation-learning-by-reinforcement-learning","title":"Imitation Learning by Reinforcement Learning","date":"2021-08-10","arxiv_id":"2108.04763","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-learning-by-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2108.04763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.04763"}},"official":{"repos":["spotify-research/il-by-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/igibson-2-0-object-centric-simulation-for","slug":"igibson-2-0-object-centric-simulation-for","title":"iGibson 2.0: Object-Centric Simulation for Robot Learning of Everyday Household Tasks","date":"2021-08-06","arxiv_id":"2108.03272","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/igibson-2-0-object-centric-simulation-for#ran","syntology_url":"https://syntology.ai/paper/2108.03272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.03272"}},"official":null}},{"url":"/paper/what-matters-in-learning-from-offline-human","slug":"what-matters-in-learning-from-offline-human","title":"What Matters in Learning from Offline Human Demonstrations for Robot Manipulation","date":"2021-08-06","arxiv_id":"2108.03298","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/what-matters-in-learning-from-offline-human#ran","syntology_url":"https://syntology.ai/paper/2108.03298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.03298"}},"official":null}},{"url":"/paper/a-pragmatic-look-at-deep-imitation-learning","slug":"a-pragmatic-look-at-deep-imitation-learning","title":"A Pragmatic Look at Deep Imitation Learning","date":"2021-08-04","arxiv_id":"2108.01867","repositories_listed":1,"syntology":null}],"record_sha256":"26d72a106a4ba13f8f346ace35be071e9d7c74549ac2657ecf07e729160a1a5a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}