{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/reinforce/papers/2","list_of":"/method/reinforce","method":"REINFORCE","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,185],"of":185,"counts":{"archive_papers_tagged":185,"with_a_code_link":80,"where_syntology_ran_a_sample":24,"not_listed_spam_title":0,"listed":185,"listed_where_code_ran":24,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":20,"every_run_a_failure_of_syntologys_instrument":4,"listed_with_a_run_with_no_instrument_failure":20,"listed_every_run_a_failure_of_syntologys_instrument":4,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/reinforce","prev":"/method/reinforce","next":null,"papers":[{"paper":"/paper/robust-dialogue-utterance-rewriting-as","slug":"robust-dialogue-utterance-rewriting-as","title":"Robust Dialogue Utterance Rewriting as Sequence Tagging","date":"2020-12-29","arxiv_id":"2012.14535","n_code_links":1,"syntology":null},{"paper":"/paper/on-emergent-systematic-generalisation-and","slug":"on-emergent-systematic-generalisation-and","title":"On (Emergent) Systematic Generalisation and Compositionality in Visual Referential Games with Straight-Through Gumbel-Softmax Estimator","date":"2020-12-19","arxiv_id":"2012.10776","n_code_links":1,"syntology":null},{"paper":"/paper/inferring-learning-rules-from-animal-decision","slug":"inferring-learning-rules-from-animal-decision","title":"Inferring learning rules from animal decision-making","date":"2020-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/taylorgan-neighbor-augmented-policy-update","slug":"taylorgan-neighbor-augmented-policy-update","title":"TaylorGAN: Neighbor-Augmented Policy Update Towards Sample-Efficient Natural Language Generation","date":"2020-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/taylorgan-neighbor-augmented-policy-update-1","slug":"taylorgan-neighbor-augmented-policy-update-1","title":"TaylorGAN: Neighbor-Augmented Policy Update for Sample-Efficient Natural Language Generation","date":"2020-11-27","arxiv_id":"2011.13527","n_code_links":1,"syntology":null},{"paper":null,"slug":"hindsight-network-credit-assignment","title":"Hindsight Network Credit Assignment","date":"2020-11-24","arxiv_id":"2011.12351","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-neural-architecture-search-for-end","slug":"efficient-neural-architecture-search-for-end","title":"Efficient Neural Architecture Search for End-to-end Speech Recognition via Straight-Through Gradients","date":"2020-11-11","arxiv_id":"2011.05649","n_code_links":1,"syntology":null},{"paper":null,"slug":"dirichlet-policies-for-reinforced-factor","title":"Dirichlet policies for reinforced factor portfolios","date":"2020-11-10","arxiv_id":"2011.05381","n_code_links":0,"syntology":null},{"paper":"/paper/guided-dialogue-policy-learning-without","slug":"guided-dialogue-policy-learning-without","title":"Guided Dialogue Policy Learning without Adversarial Learning in the Loop","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/pomo-policy-optimization-with-multiple-optima","slug":"pomo-policy-optimization-with-multiple-optima","title":"POMO: Policy Optimization with Multiple Optima for Reinforcement Learning","date":"2020-10-30","arxiv_id":"2010.16011","n_code_links":3,"syntology":{"ran":1,"of":8,"n_ran_checked":1,"n_instrument":0,"unverified":7,"pointer_only":8,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["yd-kwon/POMO"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"sample-efficient-reinforcement-learning-with-1","title":"Sample Efficient Reinforcement Learning with REINFORCE","date":"2020-10-22","arxiv_id":"2010.11364","n_code_links":0,"syntology":null},{"paper":"/paper/every-hidden-unit-maximizing-output-weights","slug":"every-hidden-unit-maximizing-output-weights","title":"Learning by Competition of Self-Interested Reinforcement Learning Agents","date":"2020-10-19","arxiv_id":"2010.09770","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-does-supernet-help-in-neural-architecture","title":"How Does Supernet Help in Neural Architecture Search?","date":"2020-10-16","arxiv_id":"2010.08219","n_code_links":0,"syntology":null},{"paper":"/paper/an-alternative-to-backpropagation-in-deep","slug":"an-alternative-to-backpropagation-in-deep","title":"MAP Propagation Algorithm: Faster Learning with a Team of Reinforcement Learning Agents","date":"2020-10-15","arxiv_id":"2010.07893","n_code_links":1,"syntology":null},{"paper":null,"slug":"variance-reduced-off-policy-memory-efficient","title":"Variance-Reduced Off-Policy Memory-Efficient Policy Search","date":"2020-09-14","arxiv_id":"2009.06548","n_code_links":0,"syntology":null},{"paper":"/paper/improving-language-generation-with-sentence","slug":"improving-language-generation-with-sentence","title":"Improving Language Generation with Sentence Coherence Objective","date":"2020-09-07","arxiv_id":"2009.06358","n_code_links":1,"syntology":null},{"paper":"/paper/confuciux-autonomous-hardware-resource","slug":"confuciux-autonomous-hardware-resource","title":"ConfuciuX: Autonomous Hardware Resource Assignment for DNN Accelerators using Reinforcement Learning","date":"2020-09-04","arxiv_id":"2009.02010","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["maestro-project/confuciux"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/stochastic-fine-grained-labeling-of-multi","slug":"stochastic-fine-grained-labeling-of-multi","title":"Stochastic Fine-grained Labeling of Multi-state Sign Glosses for Continuous Sign Language Recognition","date":"2020-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"an-operator-view-of-policy-gradient-methods","title":"An operator view of policy gradient methods","date":"2020-06-19","arxiv_id":"2006.11266","n_code_links":0,"syntology":null},{"paper":"/paper/model-based-adversarial-meta-reinforcement","slug":"model-based-adversarial-meta-reinforcement","title":"Model-based Adversarial Meta-Reinforcement Learning","date":"2020-06-16","arxiv_id":"2006.08875","n_code_links":1,"syntology":null},{"paper":null,"slug":"autohas-differentiable-hyper-parameter-and","title":"AutoHAS: Efficient Hyperparameter and Architecture Search","date":"2020-06-05","arxiv_id":"2006.03656","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-importance-of-prior-knowledge-in-precise","title":"The Importance of Prior Knowledge in Precise Multimodal Prediction","date":"2020-06-04","arxiv_id":"2006.02636","n_code_links":0,"syntology":null},{"paper":"/paper/learning-optimal-environments-using-projected","slug":"learning-optimal-environments-using-projected","title":"Jointly Learning Environments and Control Policies with Projected Stochastic Gradient Ascent","date":"2020-06-02","arxiv_id":"2006.01738","n_code_links":1,"syntology":null},{"paper":"/paper/angle-based-search-space-shrinking-for-neural","slug":"angle-based-search-space-shrinking-for-neural","title":"Angle-based Search Space Shrinking for Neural Architecture Search","date":"2020-04-28","arxiv_id":"2004.13431","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"attention-routing-track-assignment-detailed","title":"Attention Routing: track-assignment detailed routing using attention-based reinforcement learning","date":"2020-04-20","arxiv_id":"2004.09473","n_code_links":0,"syntology":null},{"paper":"/paper/guided-dialog-policy-learning-without","slug":"guided-dialog-policy-learning-without","title":"Guided Dialog Policy Learning without Adversarial Learning in the Loop","date":"2020-04-07","arxiv_id":"2004.03267","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cszmli/dp-without-adv"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/a-better-variant-of-self-critical-sequence","slug":"a-better-variant-of-self-critical-sequence","title":"A Better Variant of Self-Critical Sequence Training","date":"2020-03-22","arxiv_id":"2003.09971","n_code_links":1,"syntology":null},{"paper":"/paper/a-hybrid-stochastic-policy-gradient-algorithm","slug":"a-hybrid-stochastic-policy-gradient-algorithm","title":"A Hybrid Stochastic Policy Gradient Algorithm for Reinforcement Learning","date":"2020-03-01","arxiv_id":"2003.00430","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["unc-optimization/ProxHSPGA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/estimating-gradients-for-discrete-random-1","slug":"estimating-gradients-for-discrete-random-1","title":"Estimating Gradients for Discrete Random Variables by Sampling without Replacement","date":"2020-02-14","arxiv_id":"2002.06043","n_code_links":1,"syntology":null},{"paper":"/paper/differentiating-the-black-box-optimization","slug":"differentiating-the-black-box-optimization","title":"Black-Box Optimization with Local Generative Surrogates","date":"2020-02-11","arxiv_id":"2002.04632","n_code_links":1,"syntology":null},{"paper":null,"slug":"unsupervised-program-synthesis-for-images","title":"Unsupervised Program Synthesis for Images By Sampling Without Replacement","date":"2020-01-27","arxiv_id":"2001.10119","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-federated-learning-through-1","title":"Robust Federated Learning Through Representation Matching and Adaptive Hyper-parameters","date":"2019-12-30","arxiv_id":"1912.13075","n_code_links":0,"syntology":null},{"paper":"/paper/unas-differentiable-architecture-search-meets","slug":"unas-differentiable-architecture-search-meets","title":"UNAS: Differentiable Architecture Search Meets Reinforcement Learning","date":"2019-12-16","arxiv_id":"1912.07651","n_code_links":1,"syntology":null},{"paper":"/paper/neural-predictor-for-neural-architecture","slug":"neural-predictor-for-neural-architecture","title":"Neural Predictor for Neural Architecture Search","date":"2019-12-02","arxiv_id":"1912.00848","n_code_links":2,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"scene-graph-based-image-retrieval-a-case","title":"Scene Graph based Image Retrieval -- A case study on the CLEVR Dataset","date":"2019-11-03","arxiv_id":"1911.00850","n_code_links":0,"syntology":null},{"paper":null,"slug":"hidden-state-guidance-improving-image","title":"Hidden State Guidance: Improving Image Captioning using An Image Conditioned Autoencoder","date":"2019-10-31","arxiv_id":"1910.14208","n_code_links":0,"syntology":null},{"paper":null,"slug":"all-action-policy-gradient-methods-a","title":"All-Action Policy Gradient Methods: A Numerical Integration Approach","date":"2019-10-21","arxiv_id":"1910.09093","n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-the-variance-of-policy-gradient","title":"Analyzing the Variance of Policy Gradient Estimators for the Linear-Quadratic Regulator","date":"2019-10-02","arxiv_id":"1910.01249","n_code_links":0,"syntology":null},{"paper":"/paper/190909902","slug":"190909902","title":"Deep Reinforcement Learning with Modulated Hebbian plus Q Network Architecture","date":"2019-09-21","arxiv_id":"1909.09902","n_code_links":1,"syntology":null},{"paper":"/paper/er-ae-differentially-private-text-generation","slug":"er-ae-differentially-private-text-generation","title":"ER-AE: Differentially Private Text Generation for Authorship Anonymization","date":"2019-07-20","arxiv_id":"1907.08736","n_code_links":2,"syntology":null},{"paper":"/paper/bridging-by-word-image-grounded-vocabulary","slug":"bridging-by-word-image-grounded-vocabulary","title":"Bridging by Word: Image Grounded Vocabulary Construction for Visual Captioning","date":"2019-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"direct-policy-gradients-direct-optimization","title":"Direct Policy Gradients: Direct Optimization of Policies in Discrete Action Spaces","date":"2019-06-14","arxiv_id":"1906.06062","n_code_links":0,"syntology":null},{"paper":"/paper/exploiting-uncertainty-of-loss-landscape-for","slug":"exploiting-uncertainty-of-loss-landscape-for","title":"Exploiting Uncertainty of Loss Landscape for Stochastic Optimization","date":"2019-05-30","arxiv_id":"1905.13200","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bsvineethiitg/adams"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"190513551","title":"Recurrent Existence Determination Through Policy Optimization","date":"2019-05-29","arxiv_id":"1905.13551","n_code_links":0,"syntology":null},{"paper":"/paper/interpretable-neural-predictions-with","slug":"interpretable-neural-predictions-with","title":"Interpretable Neural Predictions with Differentiable Binary Variables","date":"2019-05-20","arxiv_id":"1905.08160","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bastings/interpretable_predictions"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/am-lfs-automl-for-loss-function-search","slug":"am-lfs-automl-for-loss-function-search","title":"AM-LFS: AutoML for Loss Function Search","date":"2019-05-17","arxiv_id":"1905.07375","n_code_links":1,"syntology":null},{"paper":"/paper/arsm-augment-reinforce-swap-merge-estimator","slug":"arsm-augment-reinforce-swap-merge-estimator","title":"ARSM: Augment-REINFORCE-Swap-Merge Estimator for Gradient Backpropagation Through Categorical Variables","date":"2019-05-04","arxiv_id":"1905.01413","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ARM-gradient/ARSM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"beyond-games-bringing-exploration-to-robots","title":"Beyond Games: Bringing Exploration to Robots in Real-world","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/posterior-regularized-reinforce-for-instance","slug":"posterior-regularized-reinforce-for-instance","title":"Posterior-regularized REINFORCE for Instance Selection in Distant Supervision","date":"2019-04-17","arxiv_id":"1904.08051","n_code_links":1,"syntology":null},{"paper":null,"slug":"generation-of-synthetic-electronic-medical","title":"Generation of Synthetic Electronic Medical Record Text","date":"2018-12-06","arxiv_id":"1812.02793","n_code_links":0,"syntology":null},{"paper":"/paper/top-k-off-policy-correction-for-a-reinforce","slug":"top-k-off-policy-correction-for-a-reinforce","title":"Top-K Off-Policy Correction for a REINFORCE Recommender System","date":"2018-12-06","arxiv_id":"1812.02353","n_code_links":1,"syntology":null},{"paper":"/paper/proxylessnas-direct-neural-architecture","slug":"proxylessnas-direct-neural-architecture","title":"ProxylessNAS: Direct Neural Architecture Search on Target Task and Hardware","date":"2018-12-02","arxiv_id":"1812.00332","n_code_links":23,"syntology":{"ran":17,"of":27,"n_ran_checked":14,"n_instrument":3,"unverified":10,"pointer_only":4,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 2 honoured, 0 violated, 12 with no contract checked; 3 where Syntology's instrument failed) · 10 unverified","official":{"repos":["MIT-HAN-LAB/ProxylessNAS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"learning-to-exploit-stability-for-3d-scene","title":"Learning to Exploit Stability for 3D Scene Parsing","date":"2018-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"translating-natural-language-to-sql-using","title":"Translating Natural Language to SQL using Pointer-Generator Networks and How Decoding Order Matters","date":"2018-11-13","arxiv_id":"1811.05303","n_code_links":0,"syntology":null},{"paper":null,"slug":"macquarie-university-at-bioasq-6b-deep","title":"Macquarie University at BioASQ 6b: Deep learning and deep reinforcement learning for query-based summarisation","date":"2018-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/learning-scheduling-algorithms-for-data","slug":"learning-scheduling-algorithms-for-data","title":"Learning Scheduling Algorithms for Data Processing Clusters","date":"2018-10-03","arxiv_id":"1810.01963","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hongzimao/decima-sim"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"a-fourier-view-of-reinforce","title":"A Fourier View of REINFORCE","date":"2018-08-12","arxiv_id":"1808.03953","n_code_links":0,"syntology":null},{"paper":"/paper/arm-augment-reinforce-merge-gradient-for","slug":"arm-augment-reinforce-merge-gradient-for","title":"ARM: Augment-REINFORCE-Merge Gradient for Stochastic Binary Networks","date":"2018-07-30","arxiv_id":"1807.11143","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-globally-optimized-object-detector","title":"Learning Globally Optimized Object Detector via Policy Gradient","date":"2018-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-reweighted-wake-sleep","slug":"revisiting-reweighted-wake-sleep","title":"Revisiting Reweighted Wake-Sleep for Models with Stochastic Control Flow","date":"2018-05-26","arxiv_id":"1805.10469","n_code_links":1,"syntology":null},{"paper":null,"slug":"regan-relaxbarinforce-based-sequence","title":"ReGAN: RE[LAX|BAR|INFORCE] based Sequence Generation using GANs","date":"2018-05-08","arxiv_id":"1805.02788","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-training-for-community-question","title":"Adversarial Training for Community Question Answer Selection Based on Multi-scale Matching","date":"2018-04-22","arxiv_id":"1804.08058","n_code_links":0,"syntology":null},{"paper":"/paper/cot-cooperative-training-for-generative","slug":"cot-cooperative-training-for-generative","title":"CoT: Cooperative Training for Generative Modeling of Discrete Data","date":"2018-04-11","arxiv_id":"1804.03782","n_code_links":2,"syntology":null},{"paper":"/paper/attention-learn-to-solve-routing-problems","slug":"attention-learn-to-solve-routing-problems","title":"Attention, Learn to Solve Routing Problems!","date":"2018-03-22","arxiv_id":"1803.08475","n_code_links":15,"syntology":{"ran":27,"of":40,"n_ran_checked":20,"n_instrument":7,"unverified":13,"pointer_only":5,"phrase":"27 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 3 honoured, 0 violated, 17 with no contract checked; 7 where Syntology's instrument failed) · 13 unverified","official":{"repos":["wouterkool/attention-tsp","wouterkool/attention-learn-to-route"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"action-dependent-control-variates-for-policy","title":"Action-dependent Control Variates for Policy Optimization via Stein Identity","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-policy-gradient-for-alternating","title":"Adversarial Policy Gradient for Alternating Markov Games","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-organize-knowledge-with-n-gram","title":"LEARNING TO ORGANIZE KNOWLEDGE WITH N-GRAM MACHINES","date":"2018-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/technical-report-for-e2e-nlg-challenge","slug":"technical-report-for-e2e-nlg-challenge","title":"Technical Report for E2E NLG Challenge","date":"2017-12-19","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/differentiable-lower-bound-for-expected-bleu","slug":"differentiable-lower-bound-for-expected-bleu","title":"Differentiable lower bound for expected BLEU score","date":"2017-12-13","arxiv_id":"1712.04708","n_code_links":2,"syntology":null},{"paper":"/paper/action-depedent-control-variates-for-policy","slug":"action-depedent-control-variates-for-policy","title":"Action-depedent Control Variates for Policy Optimization via Stein's Identity","date":"2017-10-30","arxiv_id":"1710.11198","n_code_links":2,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"energy-efficient-amortized-inference-with","title":"Energy-efficient Amortized Inference with Cascaded Deep Classifiers","date":"2017-10-10","arxiv_id":"1710.03368","n_code_links":0,"syntology":null},{"paper":null,"slug":"user-driven-mobile-robot-storyboarding","title":"Rapid Probabilistic Interest Learning from Domain-Specific Pairwise Image Comparisons","date":"2017-06-19","arxiv_id":"1706.05850","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-hard-alignments-with-variational","title":"Learning Hard Alignments with Variational Inference","date":"2017-05-16","arxiv_id":"1705.05524","n_code_links":0,"syntology":null},{"paper":"/paper/inferring-and-executing-programs-for-visual","slug":"inferring-and-executing-programs-for-visual","title":"Inferring and Executing Programs for Visual Reasoning","date":"2017-05-10","arxiv_id":"1705.03633","n_code_links":5,"syntology":{"ran":4,"of":6,"n_ran_checked":1,"n_instrument":3,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["facebookresearch/clevr-iep"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"stein-variational-policy-gradient","title":"Stein Variational Policy Gradient","date":"2017-04-07","arxiv_id":"1704.02399","n_code_links":0,"syntology":null},{"paper":"/paper/rebar-low-variance-unbiased-gradient","slug":"rebar-low-variance-unbiased-gradient","title":"REBAR: Low-variance, unbiased gradient estimates for discrete latent variable models","date":"2017-03-21","arxiv_id":"1703.07370","n_code_links":3,"syntology":null},{"paper":null,"slug":"neural-symbolic-machines-learning-semantic","title":"Neural Symbolic Machines: Learning Semantic Parsers on Freebase with Weak Supervision (Short Version)","date":"2016-12-04","arxiv_id":"1612.01197","n_code_links":0,"syntology":null},{"paper":"/paper/self-critical-sequence-training-for-image","slug":"self-critical-sequence-training-for-image","title":"Self-critical Sequence Training for Image Captioning","date":"2016-12-02","arxiv_id":"1612.00563","n_code_links":31,"syntology":{"ran":12,"of":13,"n_ran_checked":5,"n_instrument":7,"unverified":1,"pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 7 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"threshold-learning-for-optimal-decision","title":"Threshold Learning for Optimal Decision Making","date":"2016-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-policy-gradient-by-exploring-under","title":"Improving Policy Gradient by Exploring Under-appreciated Rewards","date":"2016-11-28","arxiv_id":"1611.09321","n_code_links":0,"syntology":null},{"paper":"/paper/neural-symbolic-machines-learning-semantic-1","slug":"neural-symbolic-machines-learning-semantic-1","title":"Neural Symbolic Machines: Learning Semantic Parsers on Freebase with Weak Supervision","date":"2016-10-31","arxiv_id":"1611.00020","n_code_links":2,"syntology":null},{"paper":null,"slug":"episodic-exploration-for-deep-deterministic","title":"Episodic Exploration for Deep Deterministic Policies: An Application to StarCraft Micromanagement Tasks","date":"2016-09-10","arxiv_id":"1609.02993","n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-learning-of-action-detection-from","slug":"end-to-end-learning-of-action-detection-from","title":"End-to-end Learning of Action Detection from Frame Glimpses in Videos","date":"2015-11-22","arxiv_id":"1511.06984","n_code_links":1,"syntology":null},{"paper":"/paper/estimating-or-propagating-gradients-through-1","slug":"estimating-or-propagating-gradients-through-1","title":"Estimating or Propagating Gradients Through Stochastic Neurons for Conditional Computation","date":"2013-08-15","arxiv_id":"1308.3432","n_code_links":2,"syntology":null},{"paper":null,"slug":"analysis-and-improvement-of-policy-gradient","title":"Analysis and Improvement of Policy Gradient Estimation","date":"2011-12-01","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"c78618ea568cdb1b0a9a805697c74e3c19a62334ff5c170e7b0e74d33780cda9","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}