{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/imitation-learning/papers/3","list_of":"/task/imitation-learning","task":"Imitation Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":22,"rows_per_page":100,"rows":[201,300],"of":2122,"counts":{"archive_papers_tagged":2122,"with_a_code_link":691,"where_syntology_ran_a_sample":234,"not_listed_spam_title":0,"listed":2122,"listed_where_code_ran":234,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":191,"every_run_a_failure_of_syntologys_instrument":43,"listed_with_a_run_with_no_instrument_failure":191,"listed_every_run_a_failure_of_syntologys_instrument":43,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/imitation-learning","prev":"/task/imitation-learning/papers/2","next":"/task/imitation-learning/papers/4","papers":[{"url":"/paper/rethinking-inverse-reinforcement-learning","slug":"rethinking-inverse-reinforcement-learning","title":"Rethinking Inverse Reinforcement Learning: from Data Alignment to Task Alignment","date":"2024-10-31","arxiv_id":"2410.23680","repositories_listed":1,"syntology":null},{"url":"/paper/openwebvoyager-building-multimodal-web-agents","slug":"openwebvoyager-building-multimodal-web-agents","title":"OpenWebVoyager: Building Multimodal Web Agents via Iterative Real-World Exploration, Feedback and Optimization","date":"2024-10-25","arxiv_id":"2410.19609","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/openwebvoyager-building-multimodal-web-agents#ran","syntology_url":"https://syntology.ai/paper/2410.19609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19609"}},"official":{"repos":["minorjerry/openwebvoyager"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforced-imitative-trajectory-planning-for","slug":"reinforced-imitative-trajectory-planning-for","title":"Reinforced Imitative Trajectory Planning for Urban Automated Driving","date":"2024-10-21","arxiv_id":"2410.15607","repositories_listed":1,"syntology":null},{"url":"/paper/diffusing-states-and-matching-scores-a-new","slug":"diffusing-states-and-matching-scores-a-new","title":"Diffusing States and Matching Scores: A New Framework for Imitation Learning","date":"2024-10-17","arxiv_id":"2410.13855","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/diffusing-states-and-matching-scores-a-new#ran","syntology_url":"https://syntology.ai/paper/2410.13855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13855"}},"official":{"repos":["ziqian2000/smiling"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-free-world-models-for-online-imitation","slug":"reward-free-world-models-for-online-imitation","title":"Reward-free World Models for Online Imitation Learning","date":"2024-10-17","arxiv_id":"2410.14081","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/reward-free-world-models-for-online-imitation#ran","syntology_url":"https://syntology.ai/paper/2410.14081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14081"}},"official":{"repos":["tobyleelsz/iqmpc"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/deformpam-data-efficient-learning-for-long","slug":"deformpam-data-efficient-learning-for-long","title":"DeformPAM: Data-Efficient Learning for Long-horizon Deformable Object Manipulation via Preference-based Action Alignment","date":"2024-10-15","arxiv_id":"2410.11584","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-leverage-demonstration-data-in","slug":"how-to-leverage-demonstration-data-in","title":"How to Leverage Demonstration Data in Alignment for Large Language Model? A Self-Imitation Learning Perspective","date":"2024-10-14","arxiv_id":"2410.10093","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-to-leverage-demonstration-data-in#ran","syntology_url":"https://syntology.ai/paper/2410.10093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10093"}},"official":{"repos":["tengxiao1/gsil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-offline-imitation-learning-via","slug":"zero-shot-offline-imitation-learning-via","title":"Zero-Shot Offline Imitation Learning via Optimal Transport","date":"2024-10-11","arxiv_id":"2410.08751","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/zero-shot-offline-imitation-learning-via#ran","syntology_url":"https://syntology.ai/paper/2410.08751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08751"}},"official":{"repos":["martius-lab/zilot"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-with-limited-actions-via","slug":"imitation-learning-with-limited-actions-via","title":"Imitation Learning with Limited Actions via Diffusion Planners and Deep Koopman Controllers","date":"2024-10-10","arxiv_id":"2410.07584","repositories_listed":1,"syntology":null},{"url":"/paper/active-fine-tuning-of-generalist-policies","slug":"active-fine-tuning-of-generalist-policies","title":"Active Multi-task Policy Fine-tuning","date":"2024-10-07","arxiv_id":"2410.05026","repositories_listed":1,"syntology":null},{"url":"/paper/divscene-benchmarking-lvlms-for-object","slug":"divscene-benchmarking-lvlms-for-object","title":"DivScene: Benchmarking LVLMs for Object Navigation with Diverse Scenes and Objects","date":"2024-10-03","arxiv_id":"2410.02730","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/divscene-benchmarking-lvlms-for-object#ran","syntology_url":"https://syntology.ai/paper/2410.02730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02730"}},"official":{"repos":["zhaowei-wang-nlp/divscene"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/relic-a-recipe-for-64k-steps-of-in-context","slug":"relic-a-recipe-for-64k-steps-of-in-context","title":"ReLIC: A Recipe for 64k Steps of In-Context Reinforcement Learning for Embodied AI","date":"2024-10-03","arxiv_id":"2410.02751","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/relic-a-recipe-for-64k-steps-of-in-context#ran","syntology_url":"https://syntology.ai/paper/2410.02751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02751"}},"official":{"repos":["aielawady/relic"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-build-by-building-your-own","slug":"learning-to-build-by-building-your-own","title":"Learning to Build by Building Your Own Instructions","date":"2024-10-01","arxiv_id":"2410.01111","repositories_listed":1,"syntology":null},{"url":"/paper/maniskill3-gpu-parallelized-robotics","slug":"maniskill3-gpu-parallelized-robotics","title":"ManiSkill3: GPU Parallelized Robotics Simulation and Rendering for Generalizable Embodied AI","date":"2024-10-01","arxiv_id":"2410.00425","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/maniskill3-gpu-parallelized-robotics#ran","syntology_url":"https://syntology.ai/paper/2410.00425","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00425"}},"official":{"repos":["haosulab/ManiSkill"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-multiple-probabilistic-decisions","slug":"learning-multiple-probabilistic-decisions","title":"Learning Multiple Probabilistic Decisions from Latent World Model in Autonomous Driving","date":"2024-09-24","arxiv_id":"2409.15730","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-incremental-learning-of","slug":"interactive-incremental-learning-of","title":"Interactive incremental learning of generalizable skills with local trajectory modulation","date":"2024-09-09","arxiv_id":"2409.05655","repositories_listed":1,"syntology":null},{"url":"/paper/robot-utility-models-general-policies-for","slug":"robot-utility-models-general-policies-for","title":"Robot Utility Models: General Policies for Zero-Shot Deployment in New Environments","date":"2024-09-09","arxiv_id":"2409.05865","repositories_listed":1,"syntology":null},{"url":"/paper/flowretrieval-flow-guided-data-retrieval-for","slug":"flowretrieval-flow-guided-data-retrieval-for","title":"FlowRetrieval: Flow-Guided Data Retrieval for Few-Shot Imitation Learning","date":"2024-08-29","arxiv_id":"2408.16944","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/flowretrieval-flow-guided-data-retrieval-for#ran","syntology_url":"https://syntology.ai/paper/2408.16944","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.16944"}},"official":null}},{"url":"/paper/mapf-gpt-imitation-learning-for-multi-agent-1","slug":"mapf-gpt-imitation-learning-for-multi-agent-1","title":"MAPF-GPT: Imitation Learning for Multi-Agent Pathfinding at Scale","date":"2024-08-29","arxiv_id":"2409.00134","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mapf-gpt-imitation-learning-for-multi-agent-1#ran","syntology_url":"https://syntology.ai/paper/2409.00134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00134"}},"official":{"repos":["cognitiveaisystems/mapf-gpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/in-context-imitation-learning-via-next-token","slug":"in-context-imitation-learning-via-next-token","title":"In-Context Imitation Learning via Next-Token Prediction","date":"2024-08-28","arxiv_id":"2408.15980","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/in-context-imitation-learning-via-next-token#ran","syntology_url":"https://syntology.ai/paper/2408.15980","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15980"}},"official":{"repos":["Max-Fu/icrt"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/re-mix-optimizing-data-mixtures-for-large","slug":"re-mix-optimizing-data-mixtures-for-large","title":"Re-Mix: Optimizing Data Mixtures for Large Scale Imitation Learning","date":"2024-08-26","arxiv_id":"2408.14037","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-of-embodied-learning-for-object","slug":"a-survey-of-embodied-learning-for-object","title":"A Survey of Embodied Learning for Object-Centric Robotic Manipulation","date":"2024-08-21","arxiv_id":"2408.11537","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-interpretable-decision-tree","slug":"optimizing-interpretable-decision-tree","title":"Optimizing Interpretable Decision Tree Policies for Reinforcement Learning","date":"2024-08-21","arxiv_id":"2408.11632","repositories_listed":1,"syntology":null},{"url":"/paper/personalized-dynamic-difficulty-adjustment","slug":"personalized-dynamic-difficulty-adjustment","title":"Personalized Dynamic Difficulty Adjustment -- Imitation Learning Meets Reinforcement Learning","date":"2024-08-13","arxiv_id":"2408.06818","repositories_listed":1,"syntology":null},{"url":"/paper/navigating-the-human-maze-real-time-robot","slug":"navigating-the-human-maze-real-time-robot","title":"Navigating the Human Maze: Real-Time Robot Pathfinding with Generative Imitation Learning","date":"2024-08-07","arxiv_id":"2408.03807","repositories_listed":1,"syntology":null},{"url":"/paper/2408-02061","slug":"2408-02061","title":"ParkingE2E: Camera-based End-to-end Parking Network, from Images to Planning","date":"2024-08-04","arxiv_id":"2408.02061","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-for-intra-day-power-grid","slug":"imitation-learning-for-intra-day-power-grid","title":"Imitation Learning for Intra-Day Power Grid Operation through Topology Actions","date":"2024-07-29","arxiv_id":"2407.19865","repositories_listed":1,"syntology":null},{"url":"/paper/pp-til-personalized-planning-for-autonomous","slug":"pp-til-personalized-planning-for-autonomous","title":"PP-TIL: Personalized Planning for Autonomous Driving with Instance-based Transfer Imitation Learning","date":"2024-07-26","arxiv_id":"2407.18569","repositories_listed":1,"syntology":null},{"url":"/paper/pategail-a-privacy-preserving-mobility","slug":"pategail-a-privacy-preserving-mobility","title":"PateGail: A Privacy-Preserving Mobility Trajectory Generator with Imitation Learning","date":"2024-07-23","arxiv_id":"2407.16729","repositories_listed":1,"syntology":null},{"url":"/paper/the-art-of-imitation-learning-long-horizon","slug":"the-art-of-imitation-learning-long-horizon","title":"The Art of Imitation: Learning Long-Horizon Manipulation Tasks from Few Demonstrations","date":"2024-07-18","arxiv_id":"2407.13432","repositories_listed":1,"syntology":null},{"url":"/paper/exciting-action-investigating-efficient","slug":"exciting-action-investigating-efficient","title":"Exciting Action: Investigating Efficient Exploration for Learning Musculoskeletal Humanoid Locomotion","date":"2024-07-16","arxiv_id":"2407.11658","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-with-artificial-neural","slug":"imitation-learning-with-artificial-neural","title":"Imitation learning with artificial neural networks for demand response with a heuristic control approach for heat pumps","date":"2024-07-16","arxiv_id":"2407.11561","repositories_listed":1,"syntology":null},{"url":"/paper/bigym-a-demo-driven-mobile-bi-manual","slug":"bigym-a-demo-driven-mobile-bi-manual","title":"BiGym: A Demo-Driven Mobile Bi-Manual Manipulation Benchmark","date":"2024-07-10","arxiv_id":"2407.07788","repositories_listed":1,"syntology":null},{"url":"/paper/green-screen-augmentation-enables-scene","slug":"green-screen-augmentation-enables-scene","title":"Green Screen Augmentation Enables Scene Generalisation in Robotic Manipulation","date":"2024-07-10","arxiv_id":"2407.07868","repositories_listed":1,"syntology":null},{"url":"/paper/dotamath-decomposition-of-thought-with-code","slug":"dotamath-decomposition-of-thought-with-code","title":"DotaMath: Decomposition of Thought with Code Assistance and Self-correction for Mathematical Reasoning","date":"2024-07-04","arxiv_id":"2407.04078","repositories_listed":1,"syntology":null},{"url":"/paper/equivariant-diffusion-policy","slug":"equivariant-diffusion-policy","title":"Equivariant Diffusion Policy","date":"2024-07-01","arxiv_id":"2407.01812","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/equivariant-diffusion-policy#ran","syntology_url":"https://syntology.ai/paper/2407.01812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01812"}},"official":{"repos":["pointW/equidiff"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ros-llm-a-ros-framework-for-embodied-ai-with","slug":"ros-llm-a-ros-framework-for-embodied-ai-with","title":"ROS-LLM: A ROS framework for embodied AI with task feedback and structured reasoning","date":"2024-06-28","arxiv_id":"2406.19741","repositories_listed":1,"syntology":null},{"url":"/paper/iterative-sizing-field-prediction-for","slug":"iterative-sizing-field-prediction-for","title":"Iterative Sizing Field Prediction for Adaptive Mesh Generation From Expert Demonstrations","date":"2024-06-20","arxiv_id":"2406.14161","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/iterative-sizing-field-prediction-for#ran","syntology_url":"https://syntology.ai/paper/2406.14161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14161"}},"official":{"repos":["NiklasFreymuth/AMBER"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visually-robust-adversarial-imitation","slug":"visually-robust-adversarial-imitation","title":"Visually Robust Adversarial Imitation Learning from Videos with Contrastive Learning","date":"2024-06-18","arxiv_id":"2407.12792","repositories_listed":1,"syntology":null},{"url":"/paper/evil-evolution-strategies-for-generalisable","slug":"evil-evolution-strategies-for-generalisable","title":"EvIL: Evolution Strategies for Generalisable Imitation Learning","date":"2024-06-15","arxiv_id":"2406.11905","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evil-evolution-strategies-for-generalisable#ran","syntology_url":"https://syntology.ai/paper/2406.11905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11905"}},"official":{"repos":["SilviaSapora/evil"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-locality-to-boost-sample","slug":"leveraging-locality-to-boost-sample","title":"Leveraging Locality to Boost Sample Efficiency in Robotic Manipulation","date":"2024-06-15","arxiv_id":"2406.10615","repositories_listed":1,"syntology":null},{"url":"/paper/bikc-keypose-conditioned-consistency-policy","slug":"bikc-keypose-conditioned-consistency-policy","title":"BiKC: Keypose-Conditioned Consistency Policy for Bimanual Robotic Manipulation","date":"2024-06-14","arxiv_id":"2406.10093","repositories_listed":1,"syntology":null},{"url":"/paper/is-value-learning-really-the-main-bottleneck","slug":"is-value-learning-really-the-main-bottleneck","title":"Is Value Learning Really the Main Bottleneck in Offline RL?","date":"2024-06-13","arxiv_id":"2406.09329","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":6,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/is-value-learning-really-the-main-bottleneck#ran","syntology_url":"https://syntology.ai/paper/2406.09329","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09329"}},"official":null}},{"url":"/paper/mail-improving-imitation-learning-with-mamba","slug":"mail-improving-imitation-learning-with-mamba","title":"MaIL: Improving Imitation Learning with Mamba","date":"2024-06-12","arxiv_id":"2406.08234","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mail-improving-imitation-learning-with-mamba#ran","syntology_url":"https://syntology.ai/paper/2406.08234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08234"}},"official":{"repos":["alrhub/mail"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/online-adaptation-for-enhancing-imitation","slug":"online-adaptation-for-enhancing-imitation","title":"Online Adaptation for Enhancing Imitation Learning Policies","date":"2024-06-07","arxiv_id":"2406.04913","repositories_listed":1,"syntology":null},{"url":"/paper/tedi-policy-temporally-entangled-diffusion","slug":"tedi-policy-temporally-entangled-diffusion","title":"Streaming Diffusion Policy: Fast Policy Synthesis with Variable Noise Diffusion Models","date":"2024-06-07","arxiv_id":"2406.04806","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tedi-policy-temporally-entangled-diffusion#ran","syntology_url":"https://syntology.ai/paper/2406.04806","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04806"}},"official":{"repos":["Streaming-Diffusion-Policy/streaming_diffusion_policy"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/phase-amplitude-reduction-based-imitation","slug":"phase-amplitude-reduction-based-imitation","title":"Phase-Amplitude Reduction-Based Imitation Learning","date":"2024-06-06","arxiv_id":"2406.03735","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-moment-matching-distillation-of","slug":"adversarial-moment-matching-distillation-of","title":"Adversarial Moment-Matching Distillation of Large Language Models","date":"2024-06-05","arxiv_id":"2406.02959","repositories_listed":1,"syntology":null},{"url":"/paper/show-don-t-tell-aligning-language-models-with","slug":"show-don-t-tell-aligning-language-models-with","title":"Aligning Language Models with Demonstrated Feedback","date":"2024-06-02","arxiv_id":"2406.00888","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-imitation-learning-key-reasoning-steps","slug":"beyond-imitation-learning-key-reasoning-steps","title":"Beyond Imitation: Learning Key Reasoning Steps from Dual Chain-of-Thoughts in Reasoning Distillation","date":"2024-05-30","arxiv_id":"2405.19737","repositories_listed":1,"syntology":null},{"url":"/paper/adr-bc-adversarial-density-weighted","slug":"adr-bc-adversarial-density-weighted","title":"Imitating from auxiliary imperfect demonstrations via Adversarial Density Weighted Regression","date":"2024-05-28","arxiv_id":"2405.20351","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-leverage-diverse-demonstrations-in","slug":"how-to-leverage-diverse-demonstrations-in","title":"How to Leverage Diverse Demonstrations in Offline Imitation Learning","date":"2024-05-24","arxiv_id":"2405.17476","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/how-to-leverage-diverse-demonstrations-in#ran","syntology_url":"https://syntology.ai/paper/2405.17476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17476"}},"official":{"repos":["hansenhua/ilid-offline-imitation-learning"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/ollie-imitation-learning-from-offline","slug":"ollie-imitation-learning-from-offline","title":"OLLIE: Imitation Learning from Offline Pretraining to Online Finetuning","date":"2024-05-24","arxiv_id":"2405.17477","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":4,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/ollie-imitation-learning-from-offline#ran","syntology_url":"https://syntology.ai/paper/2405.17477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17477"}},"official":{"repos":["hansenhua/ollie-offline-to-online-imitation-learning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/sub-goal-distillation-a-method-to-improve","slug":"sub-goal-distillation-a-method-to-improve","title":"Sub-goal Distillation: A Method to Improve Small Language Agents","date":"2024-05-04","arxiv_id":"2405.02749","repositories_listed":1,"syntology":null},{"url":"/paper/guiding-attention-in-end-to-end-driving","slug":"guiding-attention-in-end-to-end-driving","title":"Guiding Attention in End-to-End Driving Models","date":"2024-04-30","arxiv_id":"2405.00242","repositories_listed":1,"syntology":null},{"url":"/paper/overcoming-knowledge-barriers-online","slug":"overcoming-knowledge-barriers-online","title":"Overcoming Knowledge Barriers: Online Imitation Learning from Observation with Pretrained World Models","date":"2024-04-29","arxiv_id":"2404.18896","repositories_listed":1,"syntology":null},{"url":"/paper/ag2manip-learning-novel-manipulation-skills","slug":"ag2manip-learning-novel-manipulation-skills","title":"Ag2Manip: Learning Novel Manipulation Skills with Agent-Agnostic Visual and Action Representations","date":"2024-04-26","arxiv_id":"2404.17521","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ag2manip-learning-novel-manipulation-skills#ran","syntology_url":"https://syntology.ai/paper/2404.17521","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.17521"}},"official":{"repos":["Xiaoyao-Li/Ag2Manip"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bootstrapping-linear-models-for-fast-online","slug":"bootstrapping-linear-models-for-fast-online","title":"Bootstrapping Linear Models for Fast Online Adaptation in Human-Agent Collaboration","date":"2024-04-16","arxiv_id":"2404.10733","repositories_listed":1,"syntology":null},{"url":"/paper/juicer-data-efficient-imitation-learning-for","slug":"juicer-data-efficient-imitation-learning-for","title":"JUICER: Data-Efficient Imitation Learning for Robotic Assembly","date":"2024-04-04","arxiv_id":"2404.03729","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/juicer-data-efficient-imitation-learning-for#ran","syntology_url":"https://syntology.ai/paper/2404.03729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03729"}},"official":{"repos":["ankile/imitation-juicer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/human-compatible-driving-partners-through","slug":"human-compatible-driving-partners-through","title":"Human-compatible driving partners through data-regularized self-play reinforcement learning","date":"2024-03-28","arxiv_id":"2403.19648","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-aware-deployment-of-pre-trained","slug":"uncertainty-aware-deployment-of-pre-trained","title":"Uncertainty-Aware Deployment of Pre-trained Language-Conditioned Imitation Learning Policies","date":"2024-03-27","arxiv_id":"2403.18222","repositories_listed":1,"syntology":null},{"url":"/paper/imitating-cost-constrained-behaviors-in","slug":"imitating-cost-constrained-behaviors-in","title":"Imitating Cost-Constrained Behaviors in Reinforcement Learning","date":"2024-03-26","arxiv_id":"2403.17456","repositories_listed":1,"syntology":null},{"url":"/paper/lasil-learner-aware-supervised-imitation","slug":"lasil-learner-aware-supervised-imitation","title":"LASIL: Learner-Aware Supervised Imitation Learning For Long-term Microscopic Traffic Simulation","date":"2024-03-26","arxiv_id":"2403.17601","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lasil-learner-aware-supervised-imitation#ran","syntology_url":"https://syntology.ai/paper/2403.17601","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17601"}},"official":{"repos":["kguo-cs/lsail"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-improvement-for-neural-combinatorial","slug":"self-improvement-for-neural-combinatorial","title":"Self-Improvement for Neural Combinatorial Optimization: Sample without Replacement, but Improvement","date":"2024-03-22","arxiv_id":"2403.15180","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-improvement-for-neural-combinatorial#ran","syntology_url":"https://syntology.ai/paper/2403.15180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15180"}},"official":{"repos":["grimmlab/gumbeldore"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-adversarial-inverse-reinforcement","slug":"rethinking-adversarial-inverse-reinforcement","title":"Rethinking Adversarial Inverse Reinforcement Learning: Policy Imitation, Transferable Reward Recovery and Algebraic Equilibrium Proof","date":"2024-03-21","arxiv_id":"2403.14593","repositories_listed":1,"syntology":null},{"url":"/paper/offline-imitation-of-badminton-player","slug":"offline-imitation-of-badminton-player","title":"Offline Imitation of Badminton Player Behavior via Experiential Contexts and Brownian Motion","date":"2024-03-19","arxiv_id":"2403.12406","repositories_listed":1,"syntology":null},{"url":"/paper/globally-stable-neural-imitation-policies","slug":"globally-stable-neural-imitation-policies","title":"Globally Stable Neural Imitation Policies","date":"2024-03-07","arxiv_id":"2403.04118","repositories_listed":1,"syntology":null},{"url":"/paper/3d-diffusion-policy","slug":"3d-diffusion-policy","title":"3D Diffusion Policy: Generalizable Visuomotor Policy Learning via Simple 3D Representations","date":"2024-03-06","arxiv_id":"2403.03954","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3d-diffusion-policy#ran","syntology_url":"https://syntology.ai/paper/2403.03954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03954"}},"official":{"repos":["YanjieZe/3D-Diffusion-Policy"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-datasets-a-toolkit-for","slug":"imitation-learning-datasets-a-toolkit-for","title":"Imitation Learning Datasets: A Toolkit For Creating Datasets, Training Agents and Benchmarking","date":"2024-03-01","arxiv_id":"2403.00550","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imitation-learning-datasets-a-toolkit-for#ran","syntology_url":"https://syntology.ai/paper/2403.00550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00550"}},"official":{"repos":["nathangavenski/il-datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/behavioral-refinement-via-interpolant-based","slug":"behavioral-refinement-via-interpolant-based","title":"Don't Start from Scratch: Behavioral Refinement via Interpolant-based Policy Diffusion","date":"2024-02-25","arxiv_id":"2402.16075","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/behavioral-refinement-via-interpolant-based#ran","syntology_url":"https://syntology.ai/paper/2402.16075","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16075"}},"official":{"repos":["clear-nus/bridger"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/himap-learning-heuristics-informed-policies","slug":"himap-learning-heuristics-informed-policies","title":"HiMAP: Learning Heuristics-Informed Policies for Large-Scale Multi-Agent Pathfinding","date":"2024-02-23","arxiv_id":"2402.15546","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/himap-learning-heuristics-informed-policies#ran","syntology_url":"https://syntology.ai/paper/2402.15546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15546"}},"official":{"repos":["kaist-silab/himap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-generative-models-for-offline-policy","slug":"deep-generative-models-for-offline-policy","title":"Deep Generative Models for Offline Policy Learning: Tutorial, Survey, and Perspectives on Future Directions","date":"2024-02-21","arxiv_id":"2402.13777","repositories_listed":1,"syntology":null},{"url":"/paper/subiq-inverse-soft-q-learning-for-offline","slug":"subiq-inverse-soft-q-learning-for-offline","title":"SPRINQL: Sub-optimal Demonstrations driven Offline Imitation Learning","date":"2024-02-20","arxiv_id":"2402.13147","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/subiq-inverse-soft-q-learning-for-offline#ran","syntology_url":"https://syntology.ai/paper/2402.13147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13147"}},"official":{"repos":["hmhuy0/SPRINQL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tiny-reinforcement-learning-for-quadruped","slug":"tiny-reinforcement-learning-for-quadruped","title":"Tiny Reinforcement Learning for Quadruped Locomotion using Decision Transformers","date":"2024-02-20","arxiv_id":"2402.13201","repositories_listed":1,"syntology":null},{"url":"/paper/prise-learning-temporal-action-abstractions","slug":"prise-learning-temporal-action-abstractions","title":"PRISE: LLM-Style Sequence Compression for Learning Temporal Action Abstractions in Control","date":"2024-02-16","arxiv_id":"2402.10450","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prise-learning-temporal-action-abstractions#ran","syntology_url":"https://syntology.ai/paper/2402.10450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10450"}},"official":{"repos":["frankzheng2022/prise"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-inverse-reinforcement-learning","slug":"hybrid-inverse-reinforcement-learning","title":"Hybrid Inverse Reinforcement Learning","date":"2024-02-13","arxiv_id":"2402.08848","repositories_listed":1,"syntology":null},{"url":"/paper/a-competition-winning-deep-reinforcement","slug":"a-competition-winning-deep-reinforcement","title":"A Competition Winning Deep Reinforcement Learning Agent in microRTS","date":"2024-02-12","arxiv_id":"2402.08112","repositories_listed":1,"syntology":null},{"url":"/paper/policy-improvement-using-language-feedback","slug":"policy-improvement-using-language-feedback","title":"Policy Improvement using Language Feedback Models","date":"2024-02-12","arxiv_id":"2402.07876","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/policy-improvement-using-language-feedback#ran","syntology_url":"https://syntology.ai/paper/2402.07876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07876"}},"official":{"repos":["vzhong/language_feedback_models"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/premier-taco-pretraining-multitask","slug":"premier-taco-pretraining-multitask","title":"Premier-TACO is a Few-Shot Policy Learner: Pretraining Multitask Representation via Temporal Action-Driven Contrastive Loss","date":"2024-02-09","arxiv_id":"2402.06187","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/premier-taco-pretraining-multitask#ran","syntology_url":"https://syntology.ai/paper/2402.06187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.06187"}},"official":{"repos":["premiertaco/premier-taco"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/difftop-differentiable-trajectory","slug":"difftop-differentiable-trajectory","title":"DiffTORI: Differentiable Trajectory Optimization for Deep Reinforcement and Imitation Learning","date":"2024-02-08","arxiv_id":"2402.05421","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/difftop-differentiable-trajectory#ran","syntology_url":"https://syntology.ai/paper/2402.05421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05421"}},"official":{"repos":["wkwan7/difftori"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/oil-ad-an-anomaly-detection-framework-for","slug":"oil-ad-an-anomaly-detection-framework-for","title":"OIL-AD: An Anomaly Detection Framework for Sequential Decision Sequences","date":"2024-02-07","arxiv_id":"2402.04567","repositories_listed":1,"syntology":null},{"url":"/paper/online-cascade-learning-for-efficient","slug":"online-cascade-learning-for-efficient","title":"Online Cascade Learning for Efficient Inference over Streams","date":"2024-02-07","arxiv_id":"2402.04513","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/online-cascade-learning-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2402.04513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04513"}},"official":{"repos":["flitternie/online_cascade_learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adaflow-imitation-learning-with-variance","slug":"adaflow-imitation-learning-with-variance","title":"AdaFlow: Imitation Learning with Variance-Adaptive Flow-Based Policies","date":"2024-02-06","arxiv_id":"2402.04292","repositories_listed":1,"syntology":null},{"url":"/paper/seabo-a-simple-search-based-method-for","slug":"seabo-a-simple-search-based-method-for","title":"SEABO: A Simple Search-Based Method for Offline Imitation Learning","date":"2024-02-06","arxiv_id":"2402.03807","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seabo-a-simple-search-based-method-for#ran","syntology_url":"https://syntology.ai/paper/2402.03807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03807"}},"official":{"repos":["dmksjfl/seabo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inverse-reinforcement-learning-by-estimating","slug":"inverse-reinforcement-learning-by-estimating","title":"Inverse Reinforcement Learning by Estimating Expertise of Demonstrators","date":"2024-02-02","arxiv_id":"2402.01886","repositories_listed":1,"syntology":null},{"url":"/paper/expert-proximity-as-surrogate-rewards-for","slug":"expert-proximity-as-surrogate-rewards-for","title":"Expert Proximity as Surrogate Rewards for Single Demonstration Imitation Learning","date":"2024-02-01","arxiv_id":"2402.01057","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/expert-proximity-as-surrogate-rewards-for#ran","syntology_url":"https://syntology.ai/paper/2402.01057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01057"}},"official":{"repos":["stanl1y/tdil"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/odice-revealing-the-mystery-of-distribution","slug":"odice-revealing-the-mystery-of-distribution","title":"ODICE: Revealing the Mystery of Distribution Correction Estimation via Orthogonal-gradient Update","date":"2024-02-01","arxiv_id":"2402.00348","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":1,"n_ran_checked":5,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/odice-revealing-the-mystery-of-distribution#ran","syntology_url":"https://syntology.ai/paper/2402.00348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00348"}},"official":{"repos":["maoliyuan/odice-pytorch"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leto-learning-constrained-visuomotor-policy","slug":"leto-learning-constrained-visuomotor-policy","title":"LeTO: Learning Constrained Visuomotor Policy with Differentiable Trajectory Optimization","date":"2024-01-30","arxiv_id":"2401.17500","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-network-effect-for-fake-news","slug":"harnessing-network-effect-for-fake-news","title":"Harnessing Network Effect for Fake News Mitigation: Selecting Debunkers via Self-Imitation Learning","date":"2024-01-28","arxiv_id":"2402.03357","repositories_listed":1,"syntology":null},{"url":"/paper/states-as-strings-as-strategies-steering","slug":"states-as-strings-as-strategies-steering","title":"Steering Language Models with Game-Theoretic Solvers","date":"2024-01-24","arxiv_id":"2402.01704","repositories_listed":1,"syntology":null},{"url":"/paper/langprop-a-code-optimization-framework-using","slug":"langprop-a-code-optimization-framework-using","title":"LangProp: A code optimization framework using Large Language Models applied to driving","date":"2024-01-18","arxiv_id":"2401.10314","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/langprop-a-code-optimization-framework-using#ran","syntology_url":"https://syntology.ai/paper/2401.10314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10314"}},"official":{"repos":["shuishida/langprop"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lpac-learnable-perception-action","slug":"lpac-learnable-perception-action","title":"LPAC: Learnable Perception-Action-Communication Loops with Applications to Coverage Control","date":"2024-01-10","arxiv_id":"2401.04855","repositories_listed":1,"syntology":null},{"url":"/paper/swaptransformer-highway-overtaking-tactical","slug":"swaptransformer-highway-overtaking-tactical","title":"SwapTransformer: highway overtaking tactical planner model via imitation learning on OSHA dataset","date":"2024-01-02","arxiv_id":"2401.01425","repositories_listed":1,"syntology":null},{"url":"/paper/ensemble-based-interactive-imitation-learning","slug":"ensemble-based-interactive-imitation-learning","title":"Agnostic Interactive Imitation Learning: New Theory and Practical Algorithms","date":"2023-12-28","arxiv_id":"2312.16860","repositories_listed":1,"syntology":null},{"url":"/paper/a-trust-region-approach-for-few-shot-sim-to","slug":"a-trust-region-approach-for-few-shot-sim-to","title":"A Conservative Approach for Few-Shot Transfer in Off-Dynamics Reinforcement Learning","date":"2023-12-24","arxiv_id":"2312.15474","repositories_listed":1,"syntology":null},{"url":"/paper/lhmanip-a-dataset-for-long-horizon-language","slug":"lhmanip-a-dataset-for-long-horizon-language","title":"LHManip: A Dataset for Long-Horizon Language-Grounded Manipulation Tasks in Cluttered Tabletop Environments","date":"2023-12-19","arxiv_id":"2312.12036","repositories_listed":1,"syntology":null},{"url":"/paper/go-dice-goal-conditioned-option-aware-offline","slug":"go-dice-goal-conditioned-option-aware-offline","title":"GO-DICE: Goal-Conditioned Option-Aware Offline Imitation Learning via Stationary Distribution Correction Estimation","date":"2023-12-17","arxiv_id":"2312.10802","repositories_listed":1,"syntology":null},{"url":"/paper/diffail-diffusion-adversarial-imitation","slug":"diffail-diffusion-adversarial-imitation","title":"DiffAIL: Diffusion Adversarial Imitation Learning","date":"2023-12-11","arxiv_id":"2312.06348","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffail-diffusion-adversarial-imitation#ran","syntology_url":"https://syntology.ai/paper/2312.06348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06348"}},"official":{"repos":["ml-group-sdu/diffail"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/backward-learning-for-goal-conditioned","slug":"backward-learning-for-goal-conditioned","title":"Backward Learning for Goal-Conditioned Policies","date":"2023-12-08","arxiv_id":"2312.05044","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":1,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/backward-learning-for-goal-conditioned#ran","syntology_url":"https://syntology.ai/paper/2312.05044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.05044"}},"official":{"repos":["hauf3n/backward-learning-for-goal-conditioned-policies"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-a-surgeon-in-the-loop-ophthalmic","slug":"toward-a-surgeon-in-the-loop-ophthalmic","title":"Toward a Surgeon-in-the-Loop Ophthalmic Robotic Apprentice using Reinforcement and Imitation Learning","date":"2023-11-29","arxiv_id":"2311.17693","repositories_listed":1,"syntology":null}],"record_sha256":"c46534fa0fb39c74348dae19422e5b658b93829c501d4fbf05e5ed7586917580","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}