{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gradient-clipping/papers/2","list_of":"/method/gradient-clipping","method":"Gradient Clipping","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,167],"of":167,"counts":{"archive_papers_tagged":167,"with_a_code_link":78,"where_syntology_ran_a_sample":28,"not_listed_spam_title":0,"listed":167,"listed_where_code_ran":28,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":22,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":22,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gradient-clipping","prev":"/method/gradient-clipping","next":null,"papers":[{"paper":"/paper/summarizing-videos-using-concentrated","slug":"summarizing-videos-using-concentrated","title":"Summarizing Videos using Concentrated Attention and Considering the Uniqueness and Diversity of the Video Frames","date":"2022-06-29","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"normalized-clipped-sgd-with-perturbation-for","title":"Normalized/Clipped SGD with Perturbation for Differentially Private Non-Convex Optimization","date":"2022-06-27","arxiv_id":"2206.13033","n_code_links":0,"syntology":null},{"paper":"/paper/envpool-a-highly-parallel-reinforcement","slug":"envpool-a-highly-parallel-reinforcement","title":"EnvPool: A Highly Parallel Reinforcement Learning Environment Execution Engine","date":"2022-06-21","arxiv_id":"2206.10558","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sail-sg/envpool"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["named_in_paper","official"]}}},{"paper":"/paper/pursuit-of-a-discriminative-representation","slug":"pursuit-of-a-discriminative-representation","title":"Pursuit of a Discriminative Representation for Multiple Subspaces via Sequential Games","date":"2022-06-18","arxiv_id":"2206.09120","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["DruvPai/MultipleSubspaceRepresentationPursuit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/disparate-impact-in-differential-privacy-from","slug":"disparate-impact-in-differential-privacy-from","title":"Disparate Impact in Differential Privacy from Gradient Misalignment","date":"2022-06-15","arxiv_id":"2206.07737","n_code_links":1,"syntology":{"ran":13,"of":20,"n_ran_checked":13,"n_instrument":0,"unverified":7,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["layer6ai-labs/fair-dp"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-scalable-hyperbolic-neural-networks","title":"Towards Scalable Hyperbolic Neural Networks using Taylor Series Approximations","date":"2022-06-07","arxiv_id":"2206.03610","n_code_links":0,"syntology":null},{"paper":null,"slug":"finite-time-analysis-of-entropy-regularized","title":"Finite-Time Analysis of Entropy-Regularized Neural Natural Actor-Critic Algorithm","date":"2022-06-02","arxiv_id":"2206.00833","n_code_links":0,"syntology":null},{"paper":null,"slug":"convergence-of-stein-variational-gradient","title":"Convergence of Stein Variational Gradient Descent under a Weaker Smoothness Condition","date":"2022-06-01","arxiv_id":"2206.00508","n_code_links":0,"syntology":null},{"paper":null,"slug":"sampling-with-attribute-related-information","title":"Efficient and Training-Free Control of Language Generation","date":"2022-05-12","arxiv_id":"2205.06036","n_code_links":0,"syntology":null},{"paper":"/paper/a-communication-efficient-distributed-3","slug":"a-communication-efficient-distributed-3","title":"A Communication-Efficient Distributed Gradient Clipping Algorithm for Training Deep Neural Networks","date":"2022-05-10","arxiv_id":"2205.05040","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mingruiliu-ml-lab/communication-efficient-local-gradient-clipping"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"semantic-exploration-from-language","title":"Semantic Exploration from Language Abstractions and Pretrained Representations","date":"2022-04-08","arxiv_id":"2204.05080","n_code_links":0,"syntology":null},{"paper":null,"slug":"mirror-descent-strikes-again-optimal","title":"Mirror Descent Strikes Again: Optimal Stochastic Convex Optimization under Infinite Noise Variance","date":"2022-02-23","arxiv_id":"2202.11632","n_code_links":0,"syntology":null},{"paper":"/paper/general-cyclical-training-of-neural-networks","slug":"general-cyclical-training-of-neural-networks","title":"General Cyclical Training of Neural Networks","date":"2022-02-17","arxiv_id":"2202.08835","n_code_links":1,"syntology":null},{"paper":"/paper/backpropagation-clipping-for-deep-learning","slug":"backpropagation-clipping-for-deep-learning","title":"Backpropagation Clipping for Deep Learning with Differential Privacy","date":"2022-02-10","arxiv_id":"2202.05089","n_code_links":1,"syntology":null},{"paper":"/paper/sparsefed-mitigating-model-poisoning-attacks","slug":"sparsefed-mitigating-model-poisoning-attacks","title":"SparseFed: Mitigating Model Poisoning Attacks in Federated Learning with Sparsification","date":"2021-12-12","arxiv_id":"2112.06274","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sparsefed/sparsefed"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/combining-global-and-local-attention-with","slug":"combining-global-and-local-attention-with","title":"Combining Global and Local Attention with Positional Encoding for Video Summarization","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/differentially-private-sgd-with-sparse-1","slug":"differentially-private-sgd-with-sparse-1","title":"Improving Differentially Private SGD via Randomly Sparsified Gradients","date":"2021-12-01","arxiv_id":"2112.00845","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-transformers-on-word","title":"A Comparative Study of Transformers on Word Sense Disambiguation","date":"2021-11-30","arxiv_id":"2111.15417","n_code_links":0,"syntology":null},{"paper":null,"slug":"softmax-gradient-tampering-decoupling-the-1","title":"Altering Backward Pass Gradients improves Convergence","date":"2021-11-24","arxiv_id":"2111.12495","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-differential-privacy-preserving-sgd-1","title":"Dynamic Differential-Privacy Preserving SGD","date":"2021-10-30","arxiv_id":"2111.00173","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-distributed-deep-reinforcement-learning","title":"A Distributed Deep Reinforcement Learning Technique for Application Placement in Edge and Fog Computing Environments","date":"2021-10-24","arxiv_id":"2110.12415","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-loss-curvature-perspective-on-training-1","title":"A Loss Curvature Perspective on Training Instability in Deep Learning","date":"2021-10-08","arxiv_id":"2110.04369","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-loss-curvature-perspective-on-training","title":"A Loss Curvature Perspective on Training Instabilities of Deep Learning Models","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"losing-less-a-loss-for-differentially-private","title":"Losing Less: A Loss for Differentially Private Deep Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/turingbench-a-benchmark-environment-for","slug":"turingbench-a-benchmark-environment-for","title":"TURINGBENCH: A Benchmark Environment for Turing Test in the Age of Neural Text Generation","date":"2021-09-27","arxiv_id":"2109.13296","n_code_links":3,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/macrpo-multi-agent-cooperative-recurrent","slug":"macrpo-multi-agent-cooperative-recurrent","title":"MACRPO: Multi-Agent Cooperative Recurrent Policy Optimization","date":"2021-09-02","arxiv_id":"2109.00882","n_code_links":1,"syntology":null},{"paper":"/paper/semantic-embedded-unsupervised-spectral","slug":"semantic-embedded-unsupervised-spectral","title":"Semantic-embedded Unsupervised Spectral Reconstruction from Single RGB Images in the Wild","date":"2021-08-15","arxiv_id":"2108.06659","n_code_links":1,"syntology":null},{"paper":null,"slug":"neuraldp-differentially-private-neural","title":"NeuralDP Differentially private neural networks by design","date":"2021-07-30","arxiv_id":"2107.14582","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-clipping-for-federated-learning","title":"Understanding Clipping for Federated Learning: Convergence and Client-Level Differential Privacy","date":"2021-06-25","arxiv_id":"2106.13673","n_code_links":0,"syntology":null},{"paper":"/paper/cross-trajectory-representation-learning-for","slug":"cross-trajectory-representation-learning-for","title":"Cross-Trajectory Representation Learning for Zero-Shot Generalization in RL","date":"2021-06-04","arxiv_id":"2106.02193","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["bmazoure/ctrl_public"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-to-relate-depth-and-semantics-for","slug":"learning-to-relate-depth-and-semantics-for","title":"Learning to Relate Depth and Semantics for Unsupervised Domain Adaptation","date":"2021-05-17","arxiv_id":"2105.07830","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["susaha/ctrl-uda"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/ctlr-wic-tsv-target-sense-verification-using","slug":"ctlr-wic-tsv-target-sense-verification-using","title":"CTLR@WiC-TSV: Target Sense Verification using Marked Inputs andPre-trained Models","date":"2021-04-30","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"stability-and-convergence-of-stochastic","title":"Stability and Convergence of Stochastic Gradient Clipping: Beyond Lipschitz Continuity and Smoothness","date":"2021-02-12","arxiv_id":"2102.06489","n_code_links":0,"syntology":null},{"paper":"/paper/high-performance-large-scale-image","slug":"high-performance-large-scale-image","title":"High-Performance Large-Scale Image Recognition Without Normalization","date":"2021-02-11","arxiv_id":"2102.06171","n_code_links":20,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["deepmind/deepmind-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"eliminating-sharp-minima-from-sgd-with","title":"Eliminating Sharp Minima from SGD with Truncated Heavy-tailed Noise","date":"2021-02-08","arxiv_id":"2102.04297","n_code_links":0,"syntology":null},{"paper":null,"slug":"robustness-threats-of-differential-privacy","title":"Robustness Threats of Differential Privacy","date":"2020-12-14","arxiv_id":"2012.07828","n_code_links":0,"syntology":null},{"paper":null,"slug":"goat-gpu-outsourcing-of-deep-learning","title":"GOAT: GPU Outsourcing of Deep Learning Training With Asynchronous Probabilistic Integrity Verification Inside Trusted Execution Environment","date":"2020-10-17","arxiv_id":"2010.08855","n_code_links":0,"syntology":null},{"paper":null,"slug":"facilitate-the-parametric-dimension-reduction","title":"Facilitate the Parametric Dimension Reduction by Gradient Clipping","date":"2020-09-30","arxiv_id":"2009.14373","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-up-differentially-private-deep","slug":"scaling-up-differentially-private-deep","title":"Scaling up Differentially Private Deep Learning with Fast Per-Example Gradient Clipping","date":"2020-09-07","arxiv_id":"2009.03106","n_code_links":2,"syntology":null},{"paper":null,"slug":"training-deep-neural-networks-without-batch","title":"Training Deep Neural Networks Without Batch Normalization","date":"2020-08-18","arxiv_id":"2008.07970","n_code_links":0,"syntology":null},{"paper":"/paper/autoclip-adaptive-gradient-clipping-for","slug":"autoclip-adaptive-gradient-clipping-for","title":"AutoClip: Adaptive Gradient Clipping for Source Separation Networks","date":"2020-07-25","arxiv_id":"2007.14469","n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-gradient-clipping-in-private","title":"Understanding Gradient Clipping in Private SGD: A Geometric Perspective","date":"2020-06-27","arxiv_id":"2006.15429","n_code_links":0,"syntology":null},{"paper":"/paper/tabert-pretraining-for-joint-understanding-of","slug":"tabert-pretraining-for-joint-understanding-of","title":"TaBERT: Pretraining for Joint Understanding of Textual and Tabular Data","date":"2020-05-17","arxiv_id":"2005.08314","n_code_links":1,"syntology":null},{"paper":"/paper/differentially-private-generation-of-small","slug":"differentially-private-generation-of-small","title":"Differentially Private Generation of Small Images","date":"2020-05-02","arxiv_id":"2005.00783","n_code_links":1,"syntology":null},{"paper":"/paper/can-gradient-clipping-mitigate-label-noise","slug":"can-gradient-clipping-mitigate-label-noise","title":"Can gradient clipping mitigate label noise?","date":"2020-05-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"understanding-generalization-in-recurrent","title":"Understanding Generalization in Recurrent Neural Networks","date":"2020-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"removing-disparate-impact-of-differentially","title":"Removing Disparate Impact of Differentially Private Stochastic Gradient Descent on Model Accuracy","date":"2020-03-08","arxiv_id":"2003.03699","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-tuning-deep-reinforcement-learning","title":"A Self-Tuning Actor-Critic Algorithm","date":"2020-02-28","arxiv_id":"2002.12928","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-unified-int8-training-for","title":"Towards Unified INT8 Training for Convolutional Neural Network","date":"2019-12-29","arxiv_id":"1912.12607","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-adam-beats-sgd-for-attention-models-1","title":"Why are Adaptive Methods Good for Attention Models?","date":"2019-12-06","arxiv_id":"1912.03194","n_code_links":0,"syntology":null},{"paper":null,"slug":"impact-importance-weighted-asynchronous-1","title":"IMPACT: Importance Weighted Asynchronous Architectures with Clipped Target Networks","date":"2019-11-30","arxiv_id":"1912.00167","n_code_links":0,"syntology":null},{"paper":"/paper/compressive-transformers-for-long-range-1","slug":"compressive-transformers-for-long-range-1","title":"Compressive Transformers for Long-Range Sequence Modelling","date":"2019-11-13","arxiv_id":"1911.05507","n_code_links":6,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":"/paper/torchbeast-a-pytorch-platform-for-distributed","slug":"torchbeast-a-pytorch-platform-for-distributed","title":"TorchBeast: A PyTorch Platform for Distributed RL","date":"2019-10-08","arxiv_id":"1910.03552","n_code_links":3,"syntology":{"ran":6,"of":9,"n_ran_checked":2,"n_instrument":4,"unverified":3,"pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["heiner/scalable_agent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/ctrl-a-conditional-transformer-language-model-1","slug":"ctrl-a-conditional-transformer-language-model-1","title":"CTRL: A Conditional Transformer Language Model for Controllable Generation","date":"2019-09-11","arxiv_id":"1909.05858","n_code_links":8,"syntology":{"ran":14,"of":14,"n_ran_checked":11,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"14 ran (of which 3 constructed an object rather than computing a result; 11 with no instrument failure: 3 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/analysis-of-gradient-clipping-and-adaptive","slug":"analysis-of-gradient-clipping-and-adaptive","title":"Why gradient clipping accelerates training: A theoretical justification for adaptivity","date":"2019-05-28","arxiv_id":"1905.11881","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["JingzhaoZhang/why-clipping-accelerates"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/differential-privacy-has-disparate-impact-on","slug":"differential-privacy-has-disparate-impact-on","title":"Differential Privacy Has Disparate Impact on Model Accuracy","date":"2019-05-28","arxiv_id":"1905.12101","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ebagdasa/differential-privacy-vs-fairness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-combining-on-off-policy-methods-for","title":"Towards Combining On-Off-Policy Methods for Real-World Applications","date":"2019-04-24","arxiv_id":"1904.10642","n_code_links":0,"syntology":null},{"paper":"/paper/feature-intertwiner-for-object-detection-1","slug":"feature-intertwiner-for-object-detection-1","title":"Feature Intertwiner for Object Detection","date":"2019-03-28","arxiv_id":"1903.11851","n_code_links":2,"syntology":null},{"paper":"/paper/classification-of-medication-related-tweets","slug":"classification-of-medication-related-tweets","title":"Classification of Medication-Related Tweets Using Stacked Bidirectional LSTMs with Context-Aware Attention","date":"2018-10-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/impala-scalable-distributed-deep-rl-with","slug":"impala-scalable-distributed-deep-rl-with","title":"IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures","date":"2018-02-05","arxiv_id":"1802.01561","n_code_links":24,"syntology":{"ran":16,"of":34,"n_ran_checked":10,"n_instrument":6,"unverified":18,"pointer_only":3,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 6 where Syntology's instrument failed) · 18 unverified","official":{"repos":["deepmind/scalable_agent"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/deep-voice-3-scaling-text-to-speech-with","slug":"deep-voice-3-scaling-text-to-speech-with","title":"Deep Voice 3: Scaling Text-to-Speech with Convolutional Sequence Learning","date":"2017-10-20","arxiv_id":"1710.07654","n_code_links":7,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/riemannian-approach-to-batch-normalization","slug":"riemannian-approach-to-batch-normalization","title":"Riemannian approach to batch normalization","date":"2017-09-27","arxiv_id":"1709.09603","n_code_links":1,"syntology":{"ran":6,"of":15,"n_ran_checked":5,"n_instrument":1,"unverified":9,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","official":{"repos":["MinhyungCho/riemannian-batch-normalization"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":"/paper/terngrad-ternary-gradients-to-reduce","slug":"terngrad-ternary-gradients-to-reduce","title":"TernGrad: Ternary Gradients to Reduce Communication in Distributed Deep Learning","date":"2017-05-22","arxiv_id":"1705.07878","n_code_links":1,"syntology":null},{"paper":"/paper/language-modeling-with-gated-convolutional","slug":"language-modeling-with-gated-convolutional","title":"Language Modeling with Gated Convolutional Networks","date":"2016-12-23","arxiv_id":"1612.08083","n_code_links":11,"syntology":null},{"paper":"/paper/improving-neural-language-models-with-a","slug":"improving-neural-language-models-with-a","title":"Improving Neural Language Models with a Continuous Cache","date":"2016-12-13","arxiv_id":"1612.04426","n_code_links":14,"syntology":null},{"paper":null,"slug":"a-comprehensive-study-of-deep-bidirectional","title":"A Comprehensive Study of Deep Bidirectional LSTM RNNs for Acoustic Modeling in Speech Recognition","date":"2016-06-22","arxiv_id":"1606.06871","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-the-inception-architecture-for","slug":"rethinking-the-inception-architecture-for","title":"Rethinking the Inception Architecture for Computer Vision","date":"2015-12-02","arxiv_id":"1512.00567","n_code_links":113,"syntology":{"ran":20,"of":26,"n_ran_checked":19,"n_instrument":1,"unverified":6,"pointer_only":5,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 1 honoured, 0 violated, 18 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":null}}],"record_sha256":"b01f0060ac179de3755f95f076bbb504cb493af75cd04b4c5d28a150b83c5b49","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}