{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/40","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":40,"pages_in_order":135,"rows_per_page":100,"rows":[3901,4000],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/39","next":"/task/reinforcement-learning-2/papers/41","papers":[{"url":"/paper/sim-to-real-reinforcement-learning-for","slug":"sim-to-real-reinforcement-learning-for","title":"Sim-to-Real Reinforcement Learning for Deformable Object Manipulation","date":"2018-06-20","arxiv_id":"1806.07851","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sim-to-real-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/1806.07851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.07851"}},"official":{"repos":["JanMatas/Rainbow_ddpg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/barc-backward-reachability-curriculum-for","slug":"barc-backward-reachability-curriculum-for","title":"BaRC: Backward Reachability Curriculum for Robotic Reinforcement Learning","date":"2018-06-16","arxiv_id":"1806.06161","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/barc-backward-reachability-curriculum-for#ran","syntology_url":"https://syntology.ai/paper/1806.06161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.06161"}},"official":{"repos":["StanfordASL/BaRC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/automated-image-data-preprocessing-with-deep","slug":"automated-image-data-preprocessing-with-deep","title":"Automated Image Data Preprocessing with Deep Reinforcement Learning","date":"2018-06-15","arxiv_id":"1806.05886","repositories_listed":1,"syntology":null},{"url":"/paper/the-potential-of-the-return-distribution-for","slug":"the-potential-of-the-return-distribution-for","title":"The Potential of the Return Distribution for Exploration in RL","date":"2018-06-11","arxiv_id":"1806.04242","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-potential-of-the-return-distribution-for#ran","syntology_url":"https://syntology.ai/paper/1806.04242","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.04242"}},"official":{"repos":["tmoer/return_distribution_exploration"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-chinese-zero","slug":"deep-reinforcement-learning-for-chinese-zero","title":"Deep Reinforcement Learning for Chinese Zero pronoun Resolution","date":"2018-06-10","arxiv_id":"1806.03711","repositories_listed":1,"syntology":null},{"url":"/paper/randomized-prior-functions-for-deep","slug":"randomized-prior-functions-for-deep","title":"Randomized Prior Functions for Deep Reinforcement Learning","date":"2018-06-08","arxiv_id":"1806.03335","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/randomized-prior-functions-for-deep#ran","syntology_url":"https://syntology.ai/paper/1806.03335","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.03335"}},"official":null}},{"url":"/paper/temporal-difference-variational-auto-encoder","slug":"temporal-difference-variational-auto-encoder","title":"Temporal Difference Variational Auto-Encoder","date":"2018-06-08","arxiv_id":"1806.03107","repositories_listed":1,"syntology":null},{"url":"/paper/deep-variational-reinforcement-learning-for","slug":"deep-variational-reinforcement-learning-for","title":"Deep Variational Reinforcement Learning for POMDPs","date":"2018-06-06","arxiv_id":"1806.02426","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-variational-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/1806.02426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02426"}},"official":{"repos":["maximilianigl/DVRL"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bindsnet-a-machine-learning-oriented-spiking","slug":"bindsnet-a-machine-learning-oriented-spiking","title":"BindsNET: A machine learning-oriented spiking neural networks library in Python","date":"2018-06-04","arxiv_id":"1806.01423","repositories_listed":1,"syntology":null},{"url":"/paper/challenges-in-high-dimensional-reinforcement","slug":"challenges-in-high-dimensional-reinforcement","title":"Challenges in High-dimensional Reinforcement Learning with Evolution Strategies","date":"2018-06-04","arxiv_id":"1806.01224","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/challenges-in-high-dimensional-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1806.01224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.01224"}},"official":{"repos":["NiMlr/High-Dim-ES-RL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/playing-atari-with-six-neurons","slug":"playing-atari-with-six-neurons","title":"Playing Atari with Six Neurons","date":"2018-06-04","arxiv_id":"1806.01363","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-of-region","slug":"deep-reinforcement-learning-of-region","title":"Deep Reinforcement Learning of Region Proposal Networks for Object Detection","date":"2018-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-continual-learning","slug":"reinforced-continual-learning","title":"Reinforced Continual Learning","date":"2018-05-31","arxiv_id":"1805.12369","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-deep-reinforcement-learning-2","slug":"sample-efficient-deep-reinforcement-learning-2","title":"Sample-Efficient Deep Reinforcement Learning via Episodic Backward Update","date":"2018-05-31","arxiv_id":"1805.12375","repositories_listed":1,"syntology":null},{"url":"/paper/supervised-policy-update-for-deep","slug":"supervised-policy-update-for-deep","title":"Supervised Policy Update for Deep Reinforcement Learning","date":"2018-05-29","arxiv_id":"1805.11706","repositories_listed":1,"syntology":null},{"url":"/paper/memory-augmented-self-play","slug":"memory-augmented-self-play","title":"Memory Augmented Self-Play","date":"2018-05-28","arxiv_id":"1805.11016","repositories_listed":1,"syntology":null},{"url":"/paper/reliability-and-learnability-of-human-bandit","slug":"reliability-and-learnability-of-human-bandit","title":"Reliability and Learnability of Human Bandit Feedback for Sequence-to-Sequence Reinforcement Learning","date":"2018-05-27","arxiv_id":"1805.10627","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reliability-and-learnability-of-human-bandit#ran","syntology_url":"https://syntology.ai/paper/1805.10627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.10627"}},"official":null}},{"url":"/paper/intelligent-trainer-for-model-based","slug":"intelligent-trainer-for-model-based","title":"Intelligent Trainer for Model-Based Reinforcement Learning","date":"2018-05-24","arxiv_id":"1805.09496","repositories_listed":1,"syntology":null},{"url":"/paper/meta-gradient-reinforcement-learning","slug":"meta-gradient-reinforcement-learning","title":"Meta-Gradient Reinforcement Learning","date":"2018-05-24","arxiv_id":"1805.09801","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/meta-gradient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1805.09801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09801"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-of-marked","slug":"deep-reinforcement-learning-of-marked","title":"Deep Reinforcement Learning of Marked Temporal Point Processes","date":"2018-05-23","arxiv_id":"1805.09360","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-coordinated-exploration-in","slug":"scalable-coordinated-exploration-in","title":"Scalable Coordinated Exploration in Concurrent Reinforcement Learning","date":"2018-05-23","arxiv_id":"1805.08948","repositories_listed":1,"syntology":null},{"url":"/paper/guided-feature-transformation-gft-a-neural","slug":"guided-feature-transformation-gft-a-neural","title":"Guided Feature Transformation (GFT): A Neural Language Grounding Module for Embodied Agents","date":"2018-05-22","arxiv_id":"1805.08329","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-maximum-entropy-inverse","slug":"multi-task-maximum-entropy-inverse","title":"Multi-task Maximum Entropy Inverse Reinforcement Learning","date":"2018-05-22","arxiv_id":"1805.08882","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-task-maximum-entropy-inverse#ran","syntology_url":"https://syntology.ai/paper/1805.08882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.08882"}},"official":{"repos":["HumanCompatibleAI/population-irl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/where-do-you-think-youre-going-inferring","slug":"where-do-you-think-youre-going-inferring","title":"Where Do You Think You're Going?: Inferring Beliefs about Dynamics from Behavior","date":"2018-05-21","arxiv_id":"1805.08010","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/where-do-you-think-youre-going-inferring#ran","syntology_url":"https://syntology.ai/paper/1805.08010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.08010"}},"official":{"repos":["rddy/isql"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-lyapunov-based-approach-to-safe","slug":"a-lyapunov-based-approach-to-safe","title":"A Lyapunov-based Approach to Safe Reinforcement Learning","date":"2018-05-20","arxiv_id":"1805.07708","repositories_listed":1,"syntology":null},{"url":"/paper/machine-teaching-for-inverse-reinforcement","slug":"machine-teaching-for-inverse-reinforcement","title":"Machine Teaching for Inverse Reinforcement Learning: Algorithms and Applications","date":"2018-05-20","arxiv_id":"1805.07687","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/machine-teaching-for-inverse-reinforcement#ran","syntology_url":"https://syntology.ai/paper/1805.07687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.07687"}},"official":{"repos":["dsbrown1331/machine-teaching-irl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safe-policy-learning-from-observations","slug":"safe-policy-learning-from-observations","title":"Constrained Policy Improvement for Safe and Efficient Reinforcement Learning","date":"2018-05-20","arxiv_id":"1805.07805","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-video-object-segmentation-for","slug":"unsupervised-video-object-segmentation-for","title":"Unsupervised Video Object Segmentation for Deep Reinforcement Learning","date":"2018-05-20","arxiv_id":"1805.07780","repositories_listed":1,"syntology":null},{"url":"/paper/do-deep-reinforcement-learning-agents-model","slug":"do-deep-reinforcement-learning-agents-model","title":"Do deep reinforcement learning agents model intentions?","date":"2018-05-15","arxiv_id":"1805.06020","repositories_listed":1,"syntology":null},{"url":"/paper/unpaired-sentiment-to-sentiment-translation-a","slug":"unpaired-sentiment-to-sentiment-translation-a","title":"Unpaired Sentiment-to-Sentiment Translation: A Cycled Reinforcement Learning Approach","date":"2018-05-14","arxiv_id":"1805.05181","repositories_listed":1,"syntology":null},{"url":"/paper/gan-q-learning","slug":"gan-q-learning","title":"GAN Q-learning","date":"2018-05-13","arxiv_id":"1805.04874","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-reinforcement-learning-for","slug":"end-to-end-reinforcement-learning-for","title":"End-to-End Reinforcement Learning for Automatic Taxonomy Induction","date":"2018-05-10","arxiv_id":"1805.04044","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/end-to-end-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/1805.04044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.04044"}},"official":{"repos":["morningmoni/TaxoRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-estimation-for-variance-reduction-in","slug":"reward-estimation-for-variance-reduction-in","title":"Reward Estimation for Variance Reduction in Deep Reinforcement Learning","date":"2018-05-09","arxiv_id":"1805.03359","repositories_listed":1,"syntology":null},{"url":"/paper/ffnet-video-fast-forwarding-via-reinforcement","slug":"ffnet-video-fast-forwarding-via-reinforcement","title":"FFNet: Video Fast-Forwarding via Reinforcement Learning","date":"2018-05-08","arxiv_id":"1805.02792","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-approach-to","slug":"a-reinforcement-learning-approach-to","title":"A Reinforcement Learning Approach to Interactive-Predictive Neural Machine Translation","date":"2018-05-03","arxiv_id":"1805.01553","repositories_listed":1,"syntology":null},{"url":"/paper/dialog-based-interactive-image-retrieval","slug":"dialog-based-interactive-image-retrieval","title":"Dialog-based Interactive Image Retrieval","date":"2018-05-01","arxiv_id":"1805.00145","repositories_listed":1,"syntology":null},{"url":"/paper/a-tree-search-algorithm-for-sequence-labeling","slug":"a-tree-search-algorithm-for-sequence-labeling","title":"A Tree Search Algorithm for Sequence Labeling","date":"2018-04-29","arxiv_id":"1804.10911","repositories_listed":1,"syntology":null},{"url":"/paper/from-credit-assignment-to-entropy","slug":"from-credit-assignment-to-entropy","title":"From Credit Assignment to Entropy Regularization: Two New Algorithms for Neural Sequence Prediction","date":"2018-04-29","arxiv_id":"1804.10974","repositories_listed":1,"syntology":null},{"url":"/paper/decoupling-dynamics-and-reward-for-transfer","slug":"decoupling-dynamics-and-reward-for-transfer","title":"Decoupling Dynamics and Reward for Transfer Learning","date":"2018-04-27","arxiv_id":"1804.10689","repositories_listed":1,"syntology":null},{"url":"/paper/crawling-in-rogues-dungeons-with-partitioned","slug":"crawling-in-rogues-dungeons-with-partitioned","title":"Crawling in Rogue's dungeons with (partitioned) A3C","date":"2018-04-23","arxiv_id":"1804.08685","repositories_listed":1,"syntology":null},{"url":"/paper/towards-symbolic-reinforcement-learning-with","slug":"towards-symbolic-reinforcement-learning-with","title":"Towards Symbolic Reinforcement Learning with Common Sense","date":"2018-04-23","arxiv_id":"1804.08597","repositories_listed":1,"syntology":null},{"url":"/paper/lipschitz-continuity-in-model-based","slug":"lipschitz-continuity-in-model-based","title":"Lipschitz Continuity in Model-based Reinforcement Learning","date":"2018-04-19","arxiv_id":"1804.07193","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-on-overfitting-in-deep-reinforcement","slug":"a-study-on-overfitting-in-deep-reinforcement","title":"A Study on Overfitting in Deep Reinforcement Learning","date":"2018-04-18","arxiv_id":"1804.06893","repositories_listed":1,"syntology":null},{"url":"/paper/dialogue-learning-with-human-teaching-and","slug":"dialogue-learning-with-human-teaching-and","title":"Dialogue Learning with Human Teaching and Feedback in End-to-End Trainable Task-Oriented Dialogue Systems","date":"2018-04-18","arxiv_id":"1804.06512","repositories_listed":1,"syntology":null},{"url":"/paper/cytonrl-an-efficient-reinforcement-learning","slug":"cytonrl-an-efficient-reinforcement-learning","title":"CytonRL: an Efficient Reinforcement Learning Open-source Toolkit Implemented in C++","date":"2018-04-14","arxiv_id":"1804.05834","repositories_listed":1,"syntology":null},{"url":"/paper/emergence-of-linguistic-communication-from","slug":"emergence-of-linguistic-communication-from","title":"Emergence of Linguistic Communication from Referential Games with Symbolic and Pixel Input","date":"2018-04-11","arxiv_id":"1804.03984","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/emergence-of-linguistic-communication-from#ran","syntology_url":"https://syntology.ai/paper/1804.03984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.03984"}},"official":null}},{"url":"/paper/market-making-via-reinforcement-learning","slug":"market-making-via-reinforcement-learning","title":"Market Making via Reinforcement Learning","date":"2018-04-11","arxiv_id":"1804.04216","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-learning-of-communications-systems","slug":"end-to-end-learning-of-communications-systems","title":"End-to-End Learning of Communications Systems Without a Channel Model","date":"2018-04-06","arxiv_id":"1804.02276","repositories_listed":1,"syntology":null},{"url":"/paper/starcraft-micromanagement-with-reinforcement","slug":"starcraft-micromanagement-with-reinforcement","title":"StarCraft Micromanagement with Reinforcement Learning and Curriculum Transfer Learning","date":"2018-04-03","arxiv_id":"1804.00810","repositories_listed":1,"syntology":null},{"url":"/paper/towards-learning-transferable-conversational","slug":"towards-learning-transferable-conversational","title":"Towards Learning Transferable Conversational Skills using Multi-dimensional Dialogue Modelling","date":"2018-03-31","arxiv_id":"1804.00146","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-video-captioning-with-multitask","slug":"end-to-end-video-captioning-with-multitask","title":"End-to-End Video Captioning with Multitask Reinforcement Learning","date":"2018-03-21","arxiv_id":"1803.07950","repositories_listed":1,"syntology":null},{"url":"/paper/look-before-you-leap-bridging-model-free-and","slug":"look-before-you-leap-bridging-model-free-and","title":"Look Before You Leap: Bridging Model-Free and Model-Based Reinforcement Learning for Planned-Ahead Vision-and-Language Navigation","date":"2018-03-21","arxiv_id":"1803.07729","repositories_listed":1,"syntology":null},{"url":"/paper/composable-deep-reinforcement-learning-for","slug":"composable-deep-reinforcement-learning-for","title":"Composable Deep Reinforcement Learning for Robotic Manipulation","date":"2018-03-19","arxiv_id":"1803.06773","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-vision-based","slug":"deep-reinforcement-learning-for-vision-based","title":"Deep Reinforcement Learning for Vision-Based Robotic Grasping: A Simulated Comparative Evaluation of Off-Policy Methods","date":"2018-02-28","arxiv_id":"1802.10264","repositories_listed":1,"syntology":null},{"url":"/paper/the-mirage-of-action-dependent-baselines-in","slug":"the-mirage-of-action-dependent-baselines-in","title":"The Mirage of Action-Dependent Baselines in Reinforcement Learning","date":"2018-02-27","arxiv_id":"1802.10031","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-mirage-of-action-dependent-baselines-in#ran","syntology_url":"https://syntology.ai/paper/1802.10031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.10031"}},"official":null}},{"url":"/paper/modeling-others-using-oneself-in-multi-agent","slug":"modeling-others-using-oneself-in-multi-agent","title":"Modeling Others using Oneself in Multi-Agent Reinforcement Learning","date":"2018-02-26","arxiv_id":"1802.09640","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-and-imitation-learning-for","slug":"reinforcement-and-imitation-learning-for","title":"Reinforcement and Imitation Learning for Diverse Visuomotor Skills","date":"2018-02-26","arxiv_id":"1802.09564","repositories_listed":1,"syntology":null},{"url":"/paper/ranking-sentences-for-extractive","slug":"ranking-sentences-for-extractive","title":"Ranking Sentences for Extractive Summarization with Reinforcement Learning","date":"2018-02-23","arxiv_id":"1802.08636","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/ranking-sentences-for-extractive#ran","syntology_url":"https://syntology.ai/paper/1802.08636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.08636"}},"official":{"repos":["shashiongithub/Refresh"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/verifying-controllers-against-adversarial","slug":"verifying-controllers-against-adversarial","title":"Verifying Controllers Against Adversarial Examples with Bayesian Optimization","date":"2018-02-23","arxiv_id":"1802.08678","repositories_listed":1,"syntology":null},{"url":"/paper/structured-control-nets-for-deep","slug":"structured-control-nets-for-deep","title":"Structured Control Nets for Deep Reinforcement Learning","date":"2018-02-22","arxiv_id":"1802.08311","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-large-scale-fleet-management-via","slug":"efficient-large-scale-fleet-management-via","title":"Efficient Collaborative Multi-Agent Deep Reinforcement Learning for Large-Scale Fleet Management","date":"2018-02-18","arxiv_id":"1802.06444","repositories_listed":1,"syntology":null},{"url":"/paper/from-gameplay-to-symbolic-reasoning-learning","slug":"from-gameplay-to-symbolic-reasoning-learning","title":"From Gameplay to Symbolic Reasoning: Learning SAT Solver Heuristics in the Style of Alpha(Go) Zero","date":"2018-02-14","arxiv_id":"1802.05340","repositories_listed":1,"syntology":null},{"url":"/paper/gep-pg-decoupling-exploration-and","slug":"gep-pg-decoupling-exploration-and","title":"GEP-PG: Decoupling Exploration and Exploitation in Deep Reinforcement Learning Algorithms","date":"2018-02-14","arxiv_id":"1802.05054","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gep-pg-decoupling-exploration-and#ran","syntology_url":"https://syntology.ai/paper/1802.05054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.05054"}},"official":{"repos":["flowersteam/geppg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-bias-span-constrained-exploration","slug":"efficient-bias-span-constrained-exploration","title":"Efficient Bias-Span-Constrained Exploration-Exploitation in Reinforcement Learning","date":"2018-02-12","arxiv_id":"1802.04020","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-model-based-deep-reinforcement","slug":"efficient-model-based-deep-reinforcement","title":"Efficient Model-Based Deep Reinforcement Learning with Variational State Tabulation","date":"2018-02-12","arxiv_id":"1802.04325","repositories_listed":1,"syntology":null},{"url":"/paper/a-critical-investigation-of-deep","slug":"a-critical-investigation-of-deep","title":"A Critical Investigation of Deep Reinforcement Learning for Navigation","date":"2018-02-07","arxiv_id":"1802.02274","repositories_listed":1,"syntology":null},{"url":"/paper/shared-autonomy-via-deep-reinforcement","slug":"shared-autonomy-via-deep-reinforcement","title":"Shared Autonomy via Deep Reinforcement Learning","date":"2018-02-06","arxiv_id":"1802.01744","repositories_listed":1,"syntology":null},{"url":"/paper/utility-decomposition-with-deep-corrections","slug":"utility-decomposition-with-deep-corrections","title":"Decomposition Methods with Deep Corrections for Reinforcement Learning","date":"2018-02-06","arxiv_id":"1802.01772","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-programming","slug":"deep-reinforcement-learning-for-programming","title":"Deep Reinforcement Learning for Programming Language Correction","date":"2018-01-31","arxiv_id":"1801.10467","repositories_listed":1,"syntology":null},{"url":"/paper/logically-constrained-reinforcement-learning","slug":"logically-constrained-reinforcement-learning","title":"Logically-Constrained Reinforcement Learning","date":"2018-01-24","arxiv_id":"1801.08099","repositories_listed":1,"syntology":null},{"url":"/paper/psychlab-a-psychology-laboratory-for-deep","slug":"psychlab-a-psychology-laboratory-for-deep","title":"Psychlab: A Psychology Laboratory for Deep Reinforcement Learning Agents","date":"2018-01-24","arxiv_id":"1801.08116","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-deep-reinforcement-learning-learn","slug":"distributed-deep-reinforcement-learning-learn","title":"Distributed Deep Reinforcement Learning: Learn how to play Atari games in 21 minutes","date":"2018-01-09","arxiv_id":"1801.02852","repositories_listed":1,"syntology":null},{"url":"/paper/competitive-multi-agent-inverse-reinforcement","slug":"competitive-multi-agent-inverse-reinforcement","title":"Competitive Multi-agent Inverse Reinforcement Learning with Sub-optimal Demonstrations","date":"2018-01-07","arxiv_id":"1801.02124","repositories_listed":1,"syntology":null},{"url":"/paper/nervenet-learning-structured-policy-with","slug":"nervenet-learning-structured-policy-with","title":"NerveNet: Learning Structured Policy with Graph Neural Networks","date":"2018-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/residual-loss-prediction-reinforcement","slug":"residual-loss-prediction-reinforcement","title":"Residual Loss Prediction: Reinforcement Learning With No Incremental Feedback","date":"2018-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/federated-control-with-hierarchical-multi","slug":"federated-control-with-hierarchical-multi","title":"Federated Control with Hierarchical Multi-Agent Deep Reinforcement Learning","date":"2017-12-22","arxiv_id":"1712.08266","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-text-generation-and-planning-for","slug":"hierarchical-text-generation-and-planning-for","title":"Hierarchical Text Generation and Planning for Strategic Dialogue","date":"2017-12-15","arxiv_id":"1712.05846","repositories_listed":1,"syntology":null},{"url":"/paper/qlbs-q-learner-in-the-black-scholes-merton","slug":"qlbs-q-learner-in-the-black-scholes-merton","title":"QLBS: Q-Learner in the Black-Scholes(-Merton) Worlds","date":"2017-12-13","arxiv_id":"1712.04609","repositories_listed":1,"syntology":null},{"url":"/paper/a-low-cost-ethics-shaping-approach-for","slug":"a-low-cost-ethics-shaping-approach-for","title":"A Low-Cost Ethics Shaping Approach for Designing Reinforcement Learning Agents","date":"2017-12-12","arxiv_id":"1712.04172","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-low-cost-ethics-shaping-approach-for#ran","syntology_url":"https://syntology.ai/paper/1712.04172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.04172"}},"official":null}},{"url":"/paper/time-limits-in-reinforcement-learning","slug":"time-limits-in-reinforcement-learning","title":"Time Limits in Reinforcement Learning","date":"2017-12-01","arxiv_id":"1712.00378","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-de-novo-drug","slug":"deep-reinforcement-learning-for-de-novo-drug","title":"Deep Reinforcement Learning for De-Novo Drug Design","date":"2017-11-29","arxiv_id":"1711.10907","repositories_listed":1,"syntology":null},{"url":"/paper/crossmodal-attentive-skill-learner","slug":"crossmodal-attentive-skill-learner","title":"Crossmodal Attentive Skill Learner","date":"2017-11-28","arxiv_id":"1711.10314","repositories_listed":1,"syntology":null},{"url":"/paper/one-shot-reinforcement-learning-for-robot","slug":"one-shot-reinforcement-learning-for-robot","title":"One-Shot Reinforcement Learning for Robot Navigation with Interactive Replay","date":"2017-11-28","arxiv_id":"1711.10137","repositories_listed":1,"syntology":null},{"url":"/paper/risk-sensitive-inverse-reinforcement-learning","slug":"risk-sensitive-inverse-reinforcement-learning","title":"Risk-sensitive Inverse Reinforcement Learning via Semi- and Non-Parametric Methods","date":"2017-11-28","arxiv_id":"1711.10055","repositories_listed":1,"syntology":null},{"url":"/paper/divide-and-conquer-reinforcement-learning","slug":"divide-and-conquer-reinforcement-learning","title":"Divide-and-Conquer Reinforcement Learning","date":"2017-11-27","arxiv_id":"1711.09874","repositories_listed":1,"syntology":null},{"url":"/paper/generative-adversarial-network-for","slug":"generative-adversarial-network-for","title":"Generative Adversarial Network for Abstractive Text Summarization","date":"2017-11-26","arxiv_id":"1711.09357","repositories_listed":1,"syntology":null},{"url":"/paper/ethical-challenges-in-data-driven-dialogue","slug":"ethical-challenges-in-data-driven-dialogue","title":"Ethical Challenges in Data-Driven Dialogue Systems","date":"2017-11-24","arxiv_id":"1711.09050","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ethical-challenges-in-data-driven-dialogue#ran","syntology_url":"https://syntology.ai/paper/1711.09050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.09050"}},"official":{"repos":["Breakend/EthicsInDialogue"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/classification-with-costly-features-using","slug":"classification-with-costly-features-using","title":"Classification with Costly Features using Deep Reinforcement Learning","date":"2017-11-20","arxiv_id":"1711.07364","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-a-machine-to-read-maps-with-deep","slug":"teaching-a-machine-to-read-maps-with-deep","title":"Teaching a Machine to Read Maps with Deep Reinforcement Learning","date":"2017-11-20","arxiv_id":"1711.07479","repositories_listed":1,"syntology":null},{"url":"/paper/leave-no-trace-learning-to-reset-for-safe-and","slug":"leave-no-trace-learning-to-reset-for-safe-and","title":"Leave no Trace: Learning to Reset for Safe and Autonomous Reinforcement Learning","date":"2017-11-18","arxiv_id":"1711.06782","repositories_listed":1,"syntology":null},{"url":"/paper/hindsight-policy-gradients","slug":"hindsight-policy-gradients","title":"Hindsight policy gradients","date":"2017-11-16","arxiv_id":"1711.06006","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hindsight-policy-gradients#ran","syntology_url":"https://syntology.ai/paper/1711.06006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.06006"}},"official":null}},{"url":"/paper/towards-the-use-of-deep-reinforcement","slug":"towards-the-use-of-deep-reinforcement","title":"Towards the Use of Deep Reinforcement Learning with Global Policy For Query-based Extractive Summarisation","date":"2017-11-10","arxiv_id":"1711.03859","repositories_listed":1,"syntology":null},{"url":"/paper/latentpoison-adversarial-attacks-on-the","slug":"latentpoison-adversarial-attacks-on-the","title":"LatentPoison - Adversarial Attacks On The Latent Space","date":"2017-11-08","arxiv_id":"1711.02879","repositories_listed":1,"syntology":null},{"url":"/paper/can-deep-reinforcement-learning-solve-erdos","slug":"can-deep-reinforcement-learning-solve-erdos","title":"Can Deep Reinforcement Learning Solve Erdos-Selfridge-Spencer Games?","date":"2017-11-07","arxiv_id":"1711.02301","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-game-theoretic-approach-to","slug":"a-unified-game-theoretic-approach-to","title":"A Unified Game-Theoretic Approach to Multiagent Reinforcement Learning","date":"2017-11-02","arxiv_id":"1711.00832","repositories_listed":1,"syntology":null},{"url":"/paper/regret-minimization-for-partially-observable","slug":"regret-minimization-for-partially-observable","title":"Regret Minimization for Partially Observable Deep Reinforcement Learning","date":"2017-10-31","arxiv_id":"1710.11424","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regret-minimization-for-partially-observable#ran","syntology_url":"https://syntology.ai/paper/1710.11424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11424"}},"official":null}},{"url":"/paper/treeqn-and-atreec-differentiable-tree","slug":"treeqn-and-atreec-differentiable-tree","title":"TreeQN and ATreeC: Differentiable Tree-Structured Models for Deep Reinforcement Learning","date":"2017-10-31","arxiv_id":"1710.11417","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/treeqn-and-atreec-differentiable-tree#ran","syntology_url":"https://syntology.ai/paper/1710.11417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11417"}},"official":{"repos":["oxwhirl/treeqn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/eigenoption-discovery-through-the-deep","slug":"eigenoption-discovery-through-the-deep","title":"Eigenoption Discovery through the Deep Successor Representation","date":"2017-10-30","arxiv_id":"1710.11089","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eigenoption-discovery-through-the-deep#ran","syntology_url":"https://syntology.ai/paper/1710.11089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11089"}},"official":null}},{"url":"/paper/predicting-head-movement-in-panoramic-video-a","slug":"predicting-head-movement-in-panoramic-video-a","title":"Predicting Head Movement in Panoramic Video: A Deep Reinforcement Learning Approach","date":"2017-10-30","arxiv_id":"1710.10755","repositories_listed":1,"syntology":null},{"url":"/paper/learning-approximate-stochastic-transition","slug":"learning-approximate-stochastic-transition","title":"Learning Approximate Stochastic Transition Models","date":"2017-10-26","arxiv_id":"1710.09718","repositories_listed":1,"syntology":null}],"record_sha256":"32aac2f2c4f55f81a0d47f8f72247d62baeb63a91955740ba78a36ecf2c576cd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}