Files
FYS-STK4155/doc/Articles/Autodiff/17-468.bbl
T
Morten Hjorth-Jensen 48bfb84b91 added articlels
2023-12-01 15:21:57 +01:00

1476 lines
64 KiB
Plaintext
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
\begin{thebibliography}{208}
\providecommand{\natexlab}[1]{#1}
\providecommand{\url}[1]{\texttt{#1}}
\expandafter\ifx\csname urlstyle\endcsname\relax
\providecommand{\doi}[1]{doi: #1}\else
\providecommand{\doi}{doi: \begingroup \urlstyle{rm}\Url}\fi
\bibitem[Abadi et~al.(2016)Abadi, Agarwal, Barham, Brevdo, Chen, Citro,
Corrado, Davis, Dean, Devin, et~al.]{abadi2016tensorflow}
Mart{\'\i}n Abadi, Ashish Agarwal, Paul Barham, Eugene Brevdo, Zhifeng Chen,
Craig Citro, Greg~S Corrado, Andy Davis, Jeffrey Dean, Matthieu Devin, et~al.
\newblock {TensorFlow}: Large-scale machine learning on heterogeneous
distributed systems.
\newblock \emph{arXiv preprint arXiv:1603.04467}, 2016.
\bibitem[Adamson and Winant(1969)]{Adamson1969}
D.~S. Adamson and C.~W. Winant.
\newblock A {SLANG} simulation of an initially strong shock wave downstream of
an infinite area change.
\newblock In \emph{Proceedings of the Conference on Applications of
Continuous-System Simulation Languages}, pages 231--40, 1969.
\bibitem[Agarwal et~al.(2016)Agarwal, Bullins, and Hazan]{agarwal2016second}
Naman Agarwal, Brian Bullins, and Elad Hazan.
\newblock Second order stochastic optimization in linear time.
\newblock Technical Report arXiv:1602.03943, arXiv preprint, 2016.
\bibitem[{Al Seyab} and Cao(2008)]{AlSeyab2008}
R.~K. {Al Seyab} and Y.~Cao.
\newblock Nonlinear system identification for predictive control using
continuous time recurrent neural networks and automatic differentiation.
\newblock \emph{Journal of Process Control}, 18\penalty0 (6):\penalty0
568--581, 2008.
\newblock \doi{10.1016/j.jprocont.2007.10.012}.
\bibitem[Amos and Kolter(2017)]{amos2017optnet}
Brandon Amos and J~Zico Kolter.
\newblock {OptNet}: Differentiable optimization as a layer in neural networks.
\newblock \emph{arXiv preprint arXiv:1703.00443}, 2017.
\bibitem[Apostolopoulou et~al.(2009)Apostolopoulou, Sotiropoulos, Livieris, and
Pintelas]{Apostolopoulou2009}
Marianna~S. Apostolopoulou, Dimitris~G. Sotiropoulos, Ioannis~E. Livieris, and
Panagiotis Pintelas.
\newblock A memoryless {BFGS} neural network training algorithm.
\newblock In \emph{7th IEEE International Conference on Industrial Informatics,
INDIN 2009}, pages 216--221, June 2009.
\newblock \doi{10.1109/INDIN.2009.5195806}.
\bibitem[Appel(1989)]{appel1989runtime}
Andrew~W Appel.
\newblock Runtime tags aren't necessary.
\newblock \emph{Lisp and Symbolic Computation}, 2\penalty0 (2):\penalty0
153--162, 1989.
\bibitem[Bahdanau et~al.(2014)Bahdanau, Cho, and Bengio]{bahdanau2014neural}
Dzmitry Bahdanau, Kyunghyun Cho, and Yoshua Bengio.
\newblock Neural machine translation by jointly learning to align and
translate.
\newblock \emph{arXiv preprint arXiv:1409.0473}, 2014.
\bibitem[Barrett and Siskind(2013)]{Barrett2013}
Daniel~Paul Barrett and Jeffrey~Mark Siskind.
\newblock {Felzenszwalb-Baum-Welch}: Event detection by changing appearance.
\newblock \emph{arXiv preprint arXiv:1306.4746}, 2013.
\bibitem[Bastien et~al.(2012)Bastien, Lamblin, Pascanu, Bergstra, Goodfellow,
Bergeron, Bouchard, Warde-Farley, and Bengio]{Bastien2012}
Frédéric Bastien, Pascal Lamblin, Razvan Pascanu, James Bergstra, Ian
Goodfellow, Arnaud Bergeron, Nicolas Bouchard, David Warde-Farley, and Yoshua
Bengio.
\newblock Theano: new features and speed improvements.
\newblock Deep Learning and Unsupervised Feature Learning NIPS 2012 Workshop,
2012.
\bibitem[Bauer(1974)]{Bauer1974}
Friedrich~L. Bauer.
\newblock Computational graphs and rounding error.
\newblock \emph{SIAM Journal on Numerical Analysis}, 11\penalty0 (1):\penalty0
87--96, 1974.
\bibitem[Baydin et~al.(2016{\natexlab{a}})Baydin, Pearlmutter, and
Siskind]{baydin2016diffsharp}
Atılım~Güneş Baydin, Barak~A. Pearlmutter, and Jeffrey~Mark Siskind.
\newblock Diffsharp: An {AD} library for {.NET} languages.
\newblock In \emph{7th International Conference on Algorithmic Differentiation,
Christ Church Oxford, UK, September 12--15, 2016}, 2016{\natexlab{a}}.
\newblock Also arXiv:1611.03423.
\bibitem[Baydin et~al.(2016{\natexlab{b}})Baydin, Pearlmutter, and
Siskind]{baydin2016tricks}
Atılım~Güneş Baydin, Barak~A. Pearlmutter, and Jeffrey~Mark Siskind.
\newblock Tricks from deep learning.
\newblock In \emph{7th International Conference on Algorithmic Differentiation,
Christ Church Oxford, UK, September 12--15, 2016}, 2016{\natexlab{b}}.
\newblock Also arXiv:1611.03777.
\bibitem[Baydin et~al.(2018)Baydin, Cornish, Rubio, Schmidt, and
Wood]{baydin2017online}
Atılım~Güneş Baydin, Robert Cornish, David~Martínez Rubio, Mark Schmidt,
and Frank Wood.
\newblock Online learning rate adaptation with hypergradient descent.
\newblock In \emph{Sixth International Conference on Learning Representations
(ICLR), Vancouver, Canada, April 30--May 3, 2018}, 2018.
\bibitem[Beda et~al.(1959)Beda, Korolev, Sukkikh, and Frolova]{Beda1959}
L.~M. Beda, L.~N. Korolev, N.~V. Sukkikh, and T.~S. Frolova.
\newblock Programs for automatic differentiation for the machine {BESM} (in
{Russian}).
\newblock Technical report, Institute for Precise Mechanics and Computation
Techniques, Academy of Science, Moscow, USSR, 1959.
\bibitem[Bell and Burke(2008)]{Bell2008}
Bradley~M. Bell and James~V. Burke.
\newblock Algorithmic differentiation of implicit functions and optimal values.
\newblock In C.~H. Bischof, H.~M. Bücker, P.~Hovland, U.~Naumann, and J.~Utke,
editors, \emph{Advances in Automatic Differentiation}, volume~64 of
\emph{Lecture Notes in Computational Science and Engineering}, pages 67--77.
Springer Berlin Heidelberg, 2008.
\newblock \doi{10.1007/978-3-540-68942-3_7}.
\bibitem[Bendtsen and Stauning(1996)]{Bendtsen1996}
Claus Bendtsen and Ole Stauning.
\newblock {FADBAD}, a flexible {C++} package for automatic differentiation.
\newblock Technical Report IMM-REP-1996-17, Department of Mathematical
Modelling, Technical University of Denmark, Lyngby, Denmark, 1996.
\bibitem[Bengio et~al.(2013)Bengio, Courville, and
Vincent]{bengio2013representation}
Yoshua Bengio, Aaron Courville, and Pascal Vincent.
\newblock Representation learning: A review and new perspectives.
\newblock \emph{{IEEE} Transactions on Pattern Analysis and Machine
Intelligence}, 35\penalty0 (8):\penalty0 1798--1828, 2013.
\bibitem[Bert and Malik(1996)]{Bert1996}
Charles~W. Bert and Moinuddin Malik.
\newblock Differential quadrature method in computational mechanics: A review.
\newblock \emph{Applied Mechanics Reviews}, 49, 1996.
\newblock \doi{10.1115/1.3101882}.
\bibitem[Berz et~al.(1996)Berz, Makino, Shamseddine, Hoffstätter, and
Wan]{Berz1996}
Martin Berz, Kyoko Makino, Khodr Shamseddine, Georg~H. Hoffstätter, and Weishi
Wan.
\newblock {COSY} {INFINITY} and its applications in nonlinear dynamics.
\newblock In M.~Berz, C.~Bischof, G.~Corliss, and A.~Griewank, editors,
\emph{Computational Differentiation: Techniques, Applications, and Tools},
pages 363--5. Society for Industrial and Applied Mathematics, Philadelphia,
PA, 1996.
\bibitem[Bischof et~al.(1996)Bischof, Carle, Corliss, Griewank, and
Hovland]{Bischof1996}
Christian Bischof, Alan Carle, George Corliss, Andreas Griewank, and Paul
Hovland.
\newblock {ADIFOR} 2.0: Automatic differentiation of {Fortran} 77 programs.
\newblock \emph{Computational Science Engineering, IEEE}, 3\penalty0
(3):\penalty0 18--32, 1996.
\newblock \doi{10.1109/99.537089}.
\bibitem[Bischof et~al.(1997)Bischof, Roh, and {Mauer-Oats}]{Bischof1997}
Christian Bischof, Lucas Roh, and Andrew {Mauer-Oats}.
\newblock {ADIC}: An extensible automatic differentiation tool for {ANSI-C}.
\newblock \emph{Software Practice and Experience}, 27\penalty0 (12):\penalty0
1427--56, 1997.
\bibitem[Bischof et~al.(2002)Bischof, Bücker, and Lang]{Bischof2002}
Christian~H. Bischof, H.~Martin Bücker, and Bruno Lang.
\newblock Automatic differentiation for computational finance.
\newblock In E.~J. Kontoghiorghes, B.~Rustem, and S.~Siokos, editors,
\emph{Computational Methods in Decision-Making, Economics and Finance},
volume~74 of \emph{Applied Optimization}, pages 297--310. Springer US, 2002.
\newblock \doi{10.1007/978-1-4757-3613-7_15}.
\bibitem[Bischof et~al.(2006)Bischof, Bücker, Rasch, Slusanschi, and
Lang]{Bischof2006}
Christian~H. Bischof, H.~Martin Bücker, Arno Rasch, Emil Slusanschi, and Bruno
Lang.
\newblock Automatic differentiation of the general-purpose computational fluid
dynamics package {FLUENT}.
\newblock \emph{Journal of Fluids Engineering}, 129\penalty0 (5):\penalty0
652--8, 2006.
\newblock \doi{10.1115/1.2720475}.
\bibitem[Bischof et~al.(2008)Bischof, Hovland, and Norris]{Bischof2008}
Christian~H. Bischof, Paul~D. Hovland, and Boyana Norris.
\newblock On the implementation of automatic differentiation tools.
\newblock \emph{Higher-Order and Symbolic Computation}, 21\penalty0
(3):\penalty0 311--31, 2008.
\newblock \doi{10.1007/s10990-008-9034-4}.
\bibitem[Boltyanskii et~al.(1960)Boltyanskii, Gamkrelidze, and
Pontryagin]{Boltyanskii-Gamkrelidze-Pontryagin-1960a}
V.~G. Boltyanskii, R.~V. Gamkrelidze, and L.~S. Pontryagin.
\newblock The theory of optimal processes {I}: The maximum principle.
\newblock \emph{Izvest. Akad. Nauk S.S.S.R. Ser. Mat.}, 24:\penalty0 3--42,
1960.
\bibitem[Bottou(2010)]{bottou2010large}
L{\'e}on Bottou.
\newblock Large-scale machine learning with stochastic gradient descent.
\newblock In \emph{Proceedings of COMPSTAT'2010}, pages 177--186. Springer,
2010.
\bibitem[Bottou and {LeCun}(1988)]{bottou-lecun-88}
{L\'eon} Bottou and Yann {LeCun}.
\newblock {SN}: A simulator for connectionist models.
\newblock In \emph{Proceedings of NeuroNimes 88}, pages 371--382, Nimes,
France, 1988.
\newblock URL \url{http://leon.bottou.org/papers/bottou-lecun-88}.
\bibitem[Bottou and LeCun(2002)]{LUSH2002}
L\'{e}on Bottou and Yann LeCun.
\newblock Lush reference manual, 2002.
\newblock URL \url{http://lush.sourceforge.net/doc.html}.
\bibitem[Bottou et~al.(2016)Bottou, Curtis, and
Nocedal]{bottou2016optimization}
L{\'e}on Bottou, Frank~E. Curtis, and Jorge Nocedal.
\newblock Optimization methods for large-scale machine learning.
\newblock \emph{arXiv preprint arXiv:1606.04838}, 2016.
\bibitem[Bottou(1998)]{Bottou1998}
Léon Bottou.
\newblock Online learning and stochastic approximations.
\newblock \emph{On-Line Learning in Neural Networks}, 17:\penalty0 9, 1998.
\bibitem[Brezinski and Zaglia(1991)]{Brezinski1991}
Claude Brezinski and M.~Redivo Zaglia.
\newblock \emph{Extrapolation Methods: Theory and Practice}.
\newblock North-Holland, 1991.
\bibitem[Bryson and Denham(1962)]{Bryson-1962a}
A.~E. Bryson and W.~F. Denham.
\newblock A steepest ascent method for solving optimum programming problems.
\newblock \emph{Journal of Applied Mechanics}, 29\penalty0 (2):\penalty0 247,
1962.
\newblock \doi{10.1115/1.3640537}.
\bibitem[Bryson and Ho(1969)]{Bryson-Ho-1969a}
Arthur~E. Bryson and Yu-Chi Ho.
\newblock \emph{Applied Optimal Control: Optimization, Estimation, and
Control}.
\newblock Blaisdell, Waltham, MA, 1969.
\bibitem[Burden and Faires(2001)]{Burden2001}
Rirchard~L. Burden and J.~Douglas Faires.
\newblock \emph{Numerical Analysis}.
\newblock Brooks/Cole, 2001.
\bibitem[Capriotti(2011)]{Capriotti2011}
Luca Capriotti.
\newblock Fast {Greeks} by algorithmic differentiation.
\newblock \emph{Journal of Computational Finance}, 14\penalty0 (3):\penalty0 3,
2011.
\bibitem[Carmichael and Sandu(1997)]{Carmichael1997}
Gregory~R. Carmichael and Adrian Sandu.
\newblock Sensitivity analysis for atmospheric chemistry models via automatic
differentiation.
\newblock \emph{Atmospheric Environment}, 31\penalty0 (3):\penalty0 475--89,
1997.
\bibitem[Carpenter et~al.(2015)Carpenter, Hoffman, Brubaker, Lee, Li, and
Betancourt]{carpenter2015stan}
Bob Carpenter, Matthew~D Hoffman, Marcus Brubaker, Daniel Lee, Peter Li, and
Michael Betancourt.
\newblock The {Stan} math library: Reverse-mode automatic differentiation in
{C++}.
\newblock \emph{arXiv preprint arXiv:1509.07164}, 2015.
\bibitem[Carpenter et~al.(2016)Carpenter, Gelman, Hoffman, Lee, Goodrich,
Betancourt, Brubaker, Guo, Li, and Riddell]{carpenter2016stan}
Bob Carpenter, Andrew Gelman, Matt Hoffman, Daniel Lee, Ben Goodrich, Michael
Betancourt, Michael~A Brubaker, Jiqiang Guo, Peter Li, and Allen Riddell.
\newblock Stan: A probabilistic programming language.
\newblock \emph{Journal of Statistical Software}, 20:\penalty0 1--37, 2016.
\bibitem[Casanova et~al.(2002)Casanova, Sharp, Final, Christianson, and
Symonds]{casanova2002application}
Daniele Casanova, Robin~S. Sharp, Mark Final, Bruce Christianson, and Pat
Symonds.
\newblock Application of automatic diffentiation to race car performance
optimisation.
\newblock In George Corliss, Christ\`{e}le Faure, Andreas Griewank, Lauren
Hasco\"{e}t, and Uwe Naumann, editors, \emph{Automatic Differentiation of
Algorithms}, pages 117--124. Springer-Verlag New York, Inc., New York, NY,
USA, 2002.
\newblock ISBN 0-387-95305-1.
\bibitem[Charpentier and Ghemires(2000)]{Charpentier2000}
Isabelle Charpentier and Mohammed Ghemires.
\newblock Efficient adjoint derivatives: Application to the meteorological
model {Meso-NH}.
\newblock \emph{Optimization Methods and Software}, 13\penalty0 (1):\penalty0
35--63, 2000.
\bibitem[Chen and Manning(2014)]{chen2014fast}
Danqi Chen and Christopher Manning.
\newblock A fast and accurate dependency parser using neural networks.
\newblock In \emph{Proceedings of the 2014 Conference on Empirical Methods in
Natural Language Processing (EMNLP)}, pages 740--750, 2014.
\bibitem[Chetlur et~al.(2014)Chetlur, Woolley, Vandermersch, Cohen, Tran,
Catanzaro, and Shelhamer]{chetlur2014cudnn}
Sharan Chetlur, Cliff Woolley, Philippe Vandermersch, Jonathan Cohen, John
Tran, Bryan Catanzaro, and Evan Shelhamer.
\newblock {cuDNN}: Efficient primitives for deep learning.
\newblock \emph{arXiv preprint arXiv:1410.0759}, 2014.
\bibitem[Chib and Greenberg(1995)]{Chib1995}
Siddhartha Chib and Edward Greenberg.
\newblock Understanding the {Metropolis-Hastings} algorithm.
\newblock \emph{The American Statistician}, 49\penalty0 (4):\penalty0 327--335,
1995.
\newblock \doi{10.1080/00031305.1995.10476177}.
\bibitem[Christianson(1994)]{christianson1994reverse}
Bruce Christianson.
\newblock Reverse accumulation and attractive fixed points.
\newblock \emph{Optimization Methods and Software}, 3\penalty0 (4):\penalty0
311--326, 1994.
\bibitem[Christianson(2012)]{Christianson2012ALN}
Bruce Christianson.
\newblock A {L}eibniz notation for automatic differentiation.
\newblock In Shaun Forth, Paul Hovland, Eric Phipps, Jean Utke, and Andrea
Walther, editors, \emph{Recent Advances in Algorithmic Differentiation},
volume~87 of \emph{Lecture Notes in Computational Science and Engineering},
pages 1--9. Springer, Berlin, 2012.
\newblock ISBN 978-3-540-68935-5.
\newblock \doi{10.1007/978-3-642-30023-3_1}.
\bibitem[Clifford(1873)]{Clifford1873}
William~K. Clifford.
\newblock Preliminary sketch of bi-quaternions.
\newblock \emph{Proceedings of the London Mathematical Society}, 4:\penalty0
381--95, 1873.
\bibitem[Cohen and Molemaker(2009)]{cohen2009fast}
J~Cohen and M~Jeroen Molemaker.
\newblock A fast double precision cfd code using cuda.
\newblock \emph{Parallel Computational Fluid Dynamics: Recent Advances and
Future Directions}, pages 414--429, 2009.
\bibitem[Collobert et~al.(2011)Collobert, Kavukcuoglu, and
Farabet]{collobert2011torch7}
Ronan Collobert, Koray Kavukcuoglu, and Cl{\'e}ment Farabet.
\newblock Torch7: A {Matlab}-like environment for machine learning.
\newblock In \emph{BigLearn, NIPS Workshop}, number EPFL-CONF-192376, 2011.
\bibitem[Corliss(1988)]{Corliss1988}
George~F. Corliss.
\newblock \emph{Application of differentiation arithmetic}, volume~19 of
\emph{Perspectives in Computing}, pages 127--48.
\newblock Academic Press, Boston, 1988.
\bibitem[Courbariaux et~al.(2015)Courbariaux, Bengio, and
David]{courbariaux2015binaryconnect}
Matthieu Courbariaux, Yoshua Bengio, and Jean-Pierre David.
\newblock Binaryconnect: Training deep neural networks with binary weights
during propagations.
\newblock In \emph{Advances in Neural Information Processing Systems}, pages
3123--3131, 2015.
\bibitem[Dalal and Triggs(2005)]{Dalal2005}
Navneet Dalal and Bill Triggs.
\newblock Histograms of oriented gradients for human detection.
\newblock In \emph{Proceedings of the 2005 IEEE Computer Society Conference on
Computer Vision and Pattern Recognition (CVPR'05)}, pages 886--93,
Washington, DC, USA, 2005. IEEE Computer Society.
\newblock \doi{10.1109/CVPR.2005.177}.
\bibitem[Dauvergne and Hasco{\"e}t(2006)]{Dauvergne2006}
Benjamin Dauvergne and Laurent Hasco{\"e}t.
\newblock The data-flow equations of checkpointing in reverse automatic
differentiation.
\newblock In V.~N. Alexandrov, G.~D. {van Albada}, P.~M.~A. Sloot, and
J.~Dongarra, editors, \emph{Computational Science ICCS 2006}, volume 3994
of \emph{Lecture Notes in Computer Science}, pages 566--73, Dauvergne, 2006.
Springer Berlin.
\bibitem[Dennis and Schnabel(1996)]{Dennis1996}
John~E. Dennis and Robert~B. Schnabel.
\newblock \emph{Numerical Methods for Unconstrained Optimization and Nonlinear
Equations}.
\newblock Classics in Applied Mathematics. Society for Industrial and Applied
Mathematics, Philadelphia, 1996.
\bibitem[Dixon(1991)]{Dixon1991}
L.~C. Dixon.
\newblock Use of automatic differentiation for calculating {Hessians} and
{Newton} steps.
\newblock In A.~Griewank and G.~F. Corliss, editors, \emph{Automatic
Differentiation of Algorithms: Theory, Implementation, and Application},
pages 114--125. SIAM, Philadelphia, PA, 1991.
\bibitem[Duane et~al.(1987)Duane, Kennedy, Pendleton, and Roweth]{Duane1987}
Simon Duane, Anthony~D. Kennedy, Brian~J. Pendleton, and Duncan Roweth.
\newblock Hybrid {Monte Carlo}.
\newblock \emph{Physics Letters B}, 195\penalty0 (2):\penalty0 216--222, 1987.
\bibitem[Duchi et~al.(2011)Duchi, Hazan, and Singer]{duchi2011adaptive}
John Duchi, Elad Hazan, and Yoram Singer.
\newblock Adaptive subgradient methods for online learning and stochastic
optimization.
\newblock \emph{Journal of Machine Learning Research}, 12\penalty0
(Jul):\penalty0 2121--2159, 2011.
\bibitem[Ekstr{\"o}m et~al.(2010)Ekstr{\"o}m, Visscher, Bast, Thorvaldsen, and
Ruud]{Ekstrom2010}
Ulf Ekstr{\"o}m, Lucas Visscher, Radovan Bast, Andreas~J. Thorvaldsen, and
Kenneth Ruud.
\newblock Arbitrary-order density functional response theory from automatic
differentiation.
\newblock \emph{Journal of Chemical Theory and Computation}, 6:\penalty0
1971--80, 2010.
\newblock \doi{10.1021/ct100117s}.
\bibitem[Eriksson et~al.(1998)Eriksson, Gulliksson, Lindström, and Åke
Wedin]{Eriksson1998}
Jerry Eriksson, Mårten Gulliksson, Per Lindström, and Per Åke Wedin.
\newblock Regularization tools for training large feed-forward neural networks
using automatic differentiation.
\newblock \emph{Optimization Methods and Software}, 10\penalty0 (1):\penalty0
49--69, 1998.
\newblock \doi{10.1080/10556789808805701}.
\bibitem[Eslami et~al.(2016)Eslami, Heess, Weber, Tassa, Szepesvari,
Kavukcuoglu, and Hinton]{eslami2016attend}
S.~M.~Ali Eslami, Nicolas Heess, Theophane Weber, Yuval Tassa, David
Szepesvari, Koray Kavukcuoglu, and Geoffrey~E. Hinton.
\newblock Attend, infer, repeat: Fast scene understanding with generative
models.
\newblock In D.~D. Lee, M.~Sugiyama, U.~V. Luxburg, I.~Guyon, and R.~Garnett,
editors, \emph{Advances in Neural Information Processing Systems 29}, pages
3225--3233. Curran Associates, Inc., 2016.
\bibitem[Finkel et~al.(2008)Finkel, Kleeman, and Manning]{Finkel2008}
Jenny~Rose Finkel, Alex Kleeman, and Christopher~D. Manning.
\newblock Efficient, feature-based, conditional random field parsing.
\newblock In \emph{Proceedings of the 46th Annual Meeting of the Association
for Computational Linguistics (ACL 2008)}, pages 959--67, 2008.
\bibitem[Fornberg(1981)]{Fornberg1981}
Bengt Fornberg.
\newblock Numerical differentiation of analytic functions.
\newblock \emph{ACM Transactions on Mathematical Software}, 7\penalty0
(4):\penalty0 512--26, 1981.
\newblock \doi{10.1145/355972.355979}.
\bibitem[Forth(2006)]{Forth2006}
Shaun~A. Forth.
\newblock An efficient overloaded implementation of forward mode automatic
differentiation in {MATLAB}.
\newblock \emph{ACM Transactions on Mathematical Software}, 32\penalty0
(2):\penalty0 195--222, 2006.
\bibitem[Forth and Evans(2002)]{forth2002aerofoil}
Shaun~A. Forth and Trevor~P. Evans.
\newblock Aerofoil optimisation via {AD} of a multigrid cell-vertex {Euler}
flow solver.
\newblock In George Corliss, Christ{\`e}le Faure, Andreas Griewank, Laurent
Hasco{\"e}t, and Uwe Naumann, editors, \emph{Automatic Differentiation of
Algorithms: From Simulation to Optimization}, pages 153--160. Springer New
York, New York, NY, 2002.
\newblock ISBN 978-1-4613-0075-5.
\newblock \doi{10.1007/978-1-4613-0075-5_17}.
\bibitem[Fourer et~al.(2002)Fourer, Gay, and Kernighan]{Fourer2002}
Robert Fourer, David~M. Gay, and Brian~W. Kernighan.
\newblock \emph{{AMPL}: A Modeling Language for Mathematical Programming}.
\newblock Duxbury Press, 2002.
\bibitem[Gay(1996)]{Gay1996}
David~M. Gay.
\newblock Automatically finding and exploiting partially separable structure in
nonlinear programming problems.
\newblock Technical report, Bell Laboratories, Murray Hill, NJ, 1996.
\bibitem[Gebremedhin et~al.(2009)Gebremedhin, Tarafdar, Pothen, and
Walther]{Gebremedhin2009}
Assefaw~H. Gebremedhin, Arijit Tarafdar, Alex Pothen, and Andrea Walther.
\newblock Efficient computation of sparse {Hessians} using coloring and
automatic differentiation.
\newblock \emph{INFORMS Journal on Computing}, 21\penalty0 (2):\penalty0
209--23, 2009.
\newblock \doi{10.1287/ijoc.1080.0286}.
\bibitem[Gebremedhin et~al.(2013)Gebremedhin, Nguyen, Patwary, and
Pothen]{gebremedhin2013colpack}
Assefaw~H Gebremedhin, Duc Nguyen, Md~Mostofa~Ali Patwary, and Alex Pothen.
\newblock {ColPack}: Software for graph coloring and related problems in
scientific computing.
\newblock \emph{ACM Transactions on Mathematical Software (TOMS)}, 40\penalty0
(1):\penalty0 1, 2013.
\bibitem[Gershman and Goodman(2014)]{gershman2014amortized}
Samuel Gershman and Noah Goodman.
\newblock Amortized inference in probabilistic reasoning.
\newblock In \emph{Proceedings of the Annual Meeting of the Cognitive Science
Society}, number~36, 2014.
\bibitem[Giering and Kaminski(1998)]{Giering1998}
Ralf Giering and Thomas Kaminski.
\newblock Recipes for adjoint code construction.
\newblock \emph{ACM Transactions on Mathematical Software}, 24:\penalty0
437--74, 1998.
\newblock \doi{10.1145/293686.293695}.
\bibitem[Gimpel et~al.(2010)Gimpel, Das, and Smith]{Gimpel2010}
Kevin Gimpel, Dipanjan Das, and Noah~A. Smith.
\newblock Distributed asynchronous online learning for natural language
processing.
\newblock In \emph{Proceedings of the Fourteenth Conference on Computational
Natural Language Learning}, CoNLL '10, pages 213--222, Stroudsburg, PA, USA,
2010. Association for Computational Linguistics.
\bibitem[Girolami and Calderhead(2011)]{Girolami2011}
Mark Girolami and Be~Calderhead.
\newblock Riemann manifold {Langevin} and {Hamiltonian} {Monte Carlo} methods.
\newblock \emph{Journal of the Royal Statistical Society: Series B (Statistical
Methodology)}, 73\penalty0 (2):\penalty0 123--214, 2011.
\bibitem[Goldberg(2016)]{goldberg2016primer}
Yoav Goldberg.
\newblock A primer on neural network models for natural language processing.
\newblock \emph{Journal of Artificial Intelligence Research}, 57:\penalty0
345--420, 2016.
\bibitem[Goodfellow et~al.(2016)Goodfellow, Bengio, and
Courville]{goodfellow2016deep}
Ian Goodfellow, Yoshua Bengio, and Aaron Courville.
\newblock \emph{Deep Learning}.
\newblock MIT Press, 2016.
\newblock \url{http://www.deeplearningbook.org}.
\bibitem[Gordon et~al.(2014)Gordon, Henzinger, Nori, and
Rajamani]{gordon2014probabilistic}
Andrew~D Gordon, Thomas~A Henzinger, Aditya~V Nori, and Sriram~K Rajamani.
\newblock Probabilistic programming.
\newblock In \emph{Proceedings of the on Future of Software Engineering}, pages
167--181. ACM, 2014.
\bibitem[Grabmeier and Kaltofen(2003)]{Grabmeier2003}
Johannes Grabmeier and Erich Kaltofen.
\newblock \emph{Computer Algebra Handbook: Foundations, Applications, Systems}.
\newblock Springer, 2003.
\bibitem[Grabner et~al.(2008)Grabner, Pock, Gross, and Kainz]{Grabner2008}
Markus Grabner, Thomas Pock, Tobias Gross, and Bernhard Kainz.
\newblock Automatic differentiation for {GPU}-accelerated {2D/3D} registration.
\newblock In C.~H. Bischof, H.~M. Bücker, P.~Hovland, U.~Naumann, and J.~Utke,
editors, \emph{Advances in Automatic Differentiation}, volume~64 of
\emph{Lecture Notes in Computational Science and Engineering}, pages
259--269. Springer Berlin Heidelberg, 2008.
\newblock \doi{10.1007/978-3-540-68942-3_23}.
\bibitem[Grathwohl et~al.(2017)Grathwohl, Choi, Wu, Roeder, and
Duvenaud]{grathwohl2017backpropagation}
Will Grathwohl, Dami Choi, Yuhuai Wu, Geoff Roeder, and David Duvenaud.
\newblock Backpropagation through the void: Optimizing control variates for
black-box gradient estimation.
\newblock \emph{arXiv preprint arXiv:1711.00123}, 2017.
\bibitem[Graves et~al.(2014)Graves, Wayne, and Danihelka]{graves2014neural}
Alex Graves, Greg Wayne, and Ivo Danihelka.
\newblock Neural {Turing} machines.
\newblock \emph{arXiv preprint arXiv:1410.5401}, 2014.
\bibitem[Graves et~al.(2016)Graves, Wayne, Reynolds, Harley, Danihelka,
Grabska-Barwi{\'n}ska, Colmenarejo, Grefenstette, Ramalho, Agapiou,
et~al.]{graves2016hybrid}
Alex Graves, Greg Wayne, Malcolm Reynolds, Tim Harley, Ivo Danihelka, Agnieszka
Grabska-Barwi{\'n}ska, Sergio~G{\'o}mez Colmenarejo, Edward Grefenstette,
Tiago Ramalho, John Agapiou, et~al.
\newblock Hybrid computing using a neural network with dynamic external memory.
\newblock \emph{Nature}, 538\penalty0 (7626):\penalty0 471--476, 2016.
\bibitem[Grefenstette et~al.(2015)Grefenstette, Hermann, Suleyman, and
Blunsom]{grefenstette2015learning}
Edward Grefenstette, Karl~Moritz Hermann, Mustafa Suleyman, and Phil Blunsom.
\newblock Learning to transduce with unbounded memory.
\newblock In \emph{Advances in Neural Information Processing Systems}, pages
1828--1836, 2015.
\bibitem[Griewank(1989)]{Griewank1989}
Andreas Griewank.
\newblock On automatic differentiation.
\newblock In M.~Iri and K.~Tanabe, editors, \emph{Mathematical Programming:
Recent Developments and Applications}, pages 83--108. Kluwer Academic
Publishers, 1989.
\bibitem[Griewank(2003)]{Griewank2003}
Andreas Griewank.
\newblock A mathematical view of automatic differentiation.
\newblock \emph{Acta Numerica}, 12:\penalty0 321--98, 2003.
\newblock \doi{10.1017/S0962492902000132}.
\bibitem[Griewank(2012)]{Griewank2012}
Andreas Griewank.
\newblock Who invented the reverse mode of differentiation?
\newblock \emph{Documenta Mathematica}, Extra Volume ISMP:\penalty0 389--400,
2012.
\bibitem[Griewank and Walther(2008)]{Griewank2008}
Andreas Griewank and Andrea Walther.
\newblock \emph{Evaluating Derivatives: Principles and Techniques of
Algorithmic Differentiation}.
\newblock Society for Industrial and Applied Mathematics, Philadelphia, 2008.
\newblock \doi{10.1137/1.9780898717761}.
\bibitem[Griewank et~al.(2012)Griewank, Kulshreshtha, and
Walther]{griewank2012numerical}
Andreas Griewank, Kshitij Kulshreshtha, and Andrea Walther.
\newblock On the numerical stability of algorithmic differentiation.
\newblock \emph{Computing}, 94\penalty0 (2-4):\penalty0 125--149, 2012.
\bibitem[Gruslys et~al.(2016)Gruslys, Munos, Danihelka, Lanctot, and
Graves]{gruslys2016memory}
Audrunas Gruslys, R{\'e}mi Munos, Ivo Danihelka, Marc Lanctot, and Alex Graves.
\newblock Memory-efficient backpropagation through time.
\newblock In \emph{Advances in Neural Information Processing Systems}, pages
4125--4133, 2016.
\bibitem[Gupta et~al.(2015)Gupta, Agrawal, Gopalakrishnan, and
Narayanan]{gupta2015deep}
Suyog Gupta, Ankur Agrawal, Kailash Gopalakrishnan, and Pritish Narayanan.
\newblock Deep learning with limited numerical precision.
\newblock In \emph{Proceedings of the 32nd International Conference on Machine
Learning (ICML-15)}, pages 1737--1746, 2015.
\bibitem[Haase et~al.(2002)Haase, Langer, Lindner, and
M{\"u}hlhuber]{haase2002optimal}
Gundolf Haase, Ulrich Langer, Ewald Lindner, and Wolfram M{\"u}hlhuber.
\newblock Optimal sizing of industrial structural mechanics problems using
{AD}.
\newblock In \emph{Automatic Differentiation of Algorithms}, pages 181--188.
Springer, 2002.
\bibitem[Hadjis et~al.(2015)Hadjis, Abuzaid, Zhang, and
R{\'e}]{hadjis2015caffe}
Stefan Hadjis, Firas Abuzaid, Ce~Zhang, and Christopher R{\'e}.
\newblock Caffe con troll: Shallow ideas to speed up deep learning.
\newblock In \emph{Proceedings of the Fourth Workshop on Data analytics in the
Cloud}, page~2. ACM, 2015.
\bibitem[Hamilton(1837)]{Hamilton1837}
William~Rowan Hamilton.
\newblock Theory of conjugate functions, or algebraic couples; with a
preliminary and elementary essay on algebra as the science of pure time.
\newblock \emph{Transactions of the Royal Irish Academy}, 17:\penalty0
293--422, 1837.
\bibitem[Hasco{\"e}t and Pascual(2013)]{Hascoet2013}
Laurent Hasco{\"e}t and Valérie Pascual.
\newblock The {Tapenade} automatic differentiation tool: principles, model, and
specification.
\newblock \emph{ACM Transactions on Mathematical Software}, 39\penalty0 (3),
2013.
\newblock \doi{10.1145/2450153.2450158}.
\bibitem[Hecht-Nielsen(1989)]{Hecht1989}
Robert Hecht-Nielsen.
\newblock Theory of the backpropagation neural network.
\newblock In \emph{International Joint Conference on Neural Networks, IJCNN
1989}, pages 593--605. IEEE, 1989.
\bibitem[Hinkins(1994)]{Hinkins1994}
Ruth~L. Hinkins.
\newblock Parallel computation of automatic differentiation applied to magnetic
field calculations.
\newblock Technical report, Lawrence Berkeley Lab., CA, 1994.
\bibitem[Hinton and Ghahramani(1997)]{hinton1997generative}
Geoffrey~E. Hinton and Zoubin Ghahramani.
\newblock Generative models for discovering sparse distributed representations.
\newblock \emph{Philosophical Transactions of the Royal Society of London B:
Biological Sciences}, 352\penalty0 (1358):\penalty0 1177--1190, 1997.
\bibitem[Hoffman and Gelman(2014)]{Hoffman2014}
Matthew~D. Hoffman and Andrew Gelman.
\newblock The no-{U}-turn sampler: Adaptively setting path lengths in
{H}amiltonian {M}onte {C}arlo.
\newblock \emph{Journal of Machine Learning Research}, 15:\penalty0 1351--1381,
2014.
\bibitem[Horn(1977)]{horn1977understanding}
Berthold K.~P. Horn.
\newblock Understanding image intensities.
\newblock \emph{Artificial Intelligence}, 8:\penalty0 201--231, 1977.
\bibitem[Horwedel et~al.(1988)Horwedel, Worley, Oblow, and Pin]{Horwedel1988}
Jim~E. Horwedel, Brian~A. Worley, E.~M. Oblow, and F.~G. Pin.
\newblock {GRESS} version 1.0 user's manual.
\newblock Technical Memorandum ORNL/TM 10835, Martin Marietta Energy Systems,
Inc., Oak Ridge National Laboratory, Oak Ridge, 1988.
\bibitem[Jerrell(1997)]{Jerrell1997}
Max~E. Jerrell.
\newblock Automatic differentiation and interval arithmetic for estimation of
disequilibrium models.
\newblock \emph{Computational Economics}, 10\penalty0 (3):\penalty0 295--316,
1997.
\bibitem[Jia et~al.(2014)Jia, Shelhamer, Donahue, Karayev, Long, Girshick,
Guadarrama, and Darrell]{jia2014caffe}
Yangqing Jia, Evan Shelhamer, Jeff Donahue, Sergey Karayev, Jonathan Long, Ross
Girshick, Sergio Guadarrama, and Trevor Darrell.
\newblock Caffe: Convolutional architecture for fast feature embedding.
\newblock In \emph{Proceedings of the 22nd ACM International Conference on
Multimedia}, pages 675--678. ACM, 2014.
\bibitem[Johnson et~al.(2016)Johnson, Duvenaud, Wiltschko, Adams, and
Datta]{johnson2016composing}
Matthew Johnson, David~K Duvenaud, Alex Wiltschko, Ryan~P Adams, and Sandeep~R
Datta.
\newblock Composing graphical models with neural networks for structured
representations and fast inference.
\newblock In \emph{Advances in Neural Information Processing Systems}, pages
2946--2954, 2016.
\bibitem[Jones et~al.(1993{\natexlab{a}})Jones, Gomard, and
Sestoft]{jones1993partial}
Neil~D Jones, Carsten~K Gomard, and Peter Sestoft.
\newblock \emph{Partial evaluation and automatic program generation}.
\newblock Peter Sestoft, 1993{\natexlab{a}}.
\bibitem[Jones and Launchbury(1991)]{jones1991unboxed}
Simon L~Peyton Jones and John Launchbury.
\newblock Unboxed values as first class citizens in a non-strict functional
language.
\newblock In \emph{Conference on Functional Programming Languages and Computer
Architecture}, pages 636--666. Springer, 1991.
\bibitem[Jones et~al.(1993{\natexlab{b}})Jones, Hall, Hammond, Partain, and
Wadler]{jones1993glasgow}
SL~Peyton Jones, Cordy Hall, Kevin Hammond, Will Partain, and Philip Wadler.
\newblock The {G}lasgow {H}askell compiler: a technical overview.
\newblock In \emph{Proc. UK Joint Framework for Information Technology (JFIT)
Technical Conference}, volume~93, 1993{\natexlab{b}}.
\bibitem[Joulin and Mikolov(2015)]{joulin2015inferring}
Armand Joulin and Tomas Mikolov.
\newblock Inferring algorithmic patterns with stack-augmented recurrent nets.
\newblock In \emph{Advances in Neural Information Processing Systems}, pages
190--198, 2015.
\bibitem[Juedes(1991)]{Juedes1991}
David~W. Juedes.
\newblock A taxonomy of automatic differentiation tools.
\newblock In A.~Griewank and G.~F. Corliss, editors, \emph{Automatic
Differentiation of Algorithms: Theory, Implementation, and Application},
pages 315--29. Society for Industrial and Applied Mathematics, Philadelphia,
PA, 1991.
\bibitem[Kingma and Ba(2015)]{kingma2015adam}
D.~Kingma and J.~Ba.
\newblock Adam: A method for stochastic optimization.
\newblock In \emph{The International Conference on Learning Representations
(ICLR), San Diego}, 2015.
\bibitem[Kingma and Welling(2014)]{kingma2014auto}
Diederik~P. Kingma and Max Welling.
\newblock Auto-encoding variational {Bayes}.
\newblock In \emph{International Conference on Learning Representations}, 2014.
\bibitem[Krizhevsky et~al.(2012)Krizhevsky, Sutskever, and
Hinton]{krizhevsky2012imagenet}
Alex Krizhevsky, Ilya Sutskever, and Geoffrey~E. Hinton.
\newblock {ImageNet} classification with deep convolutional neural networks.
\newblock In \emph{Advances in Neural Information Processing Systems}, pages
1097--1105, 2012.
\bibitem[Kubo and Iri(1990)]{Kubo1990}
K.~Kubo and M.~Iri.
\newblock {PADRE2}, version 1---user's manual.
\newblock Research Memorandum RMI 90-01, Department of Mathematical Engineering
and Information Physics, University of Tokyo, Tokyo, 1990.
\bibitem[Kucukelbir et~al.(2017)Kucukelbir, Tran, Ranganath, Gelman, and
Blei]{kucukelbir2017automatic}
Alp Kucukelbir, Dustin Tran, Rajesh Ranganath, Andrew Gelman, and David~M.
Blei.
\newblock Automatic differentiation variational inference.
\newblock \emph{Journal of Machine Learning Research}, 18\penalty0
(14):\penalty0 1--45, 2017.
\bibitem[Kulkarni et~al.(2015)Kulkarni, Kohli, Tenenbaum, and
Mansinghka]{kulkarni2015picture}
Tejas~D. Kulkarni, Pushmeet Kohli, Joshua~B. Tenenbaum, and Vikash Mansinghka.
\newblock Picture: A probabilistic programming language for scene perception.
\newblock In \emph{The {IEEE} Conference on Computer Vision and Pattern
Recognition ({CVPR})}, June 2015.
\bibitem[Kumar et~al.(2016)Kumar, Irsoy, Ondruska, Iyyer, Bradbury, Gulrajani,
Zhong, Paulus, and Socher]{pmlr-v48-kumar16}
Ankit Kumar, Ozan Irsoy, Peter Ondruska, Mohit Iyyer, James Bradbury, Ishaan
Gulrajani, Victor Zhong, Romain Paulus, and Richard Socher.
\newblock Ask me anything: Dynamic memory networks for natural language
processing.
\newblock In Maria~Florina Balcan and Kilian~Q. Weinberger, editors,
\emph{Proceedings of The 33rd International Conference on Machine Learning},
volume~48 of \emph{Proceedings of Machine Learning Research}, pages
1378--1387, New York, New York, USA, 20--22 Jun 2016. PMLR.
\bibitem[Lawson(1971)]{Lawson1971}
C.~L. Lawson.
\newblock Computing derivatives using {W}-arithmetic and {U}-arithmetic.
\newblock Internal Computing Memorandum CM-286, Jet Propulsion Laboratory,
Pasadena, CA, 1971.
\bibitem[Le et~al.(2017)Le, Baydin, and Wood]{le2016inference}
Tuan~Anh Le, Atılım~Güneş Baydin, and Frank Wood.
\newblock Inference compilation and universal probabilistic programming.
\newblock In \emph{Proceedings of the 20th International Conference on
Artificial Intelligence and Statistics (AISTATS)}, volume~54 of
\emph{Proceedings of Machine Learning Research}, pages 1338--1348, Fort
Lauderdale, FL, USA, 2017. PMLR.
\bibitem[LeCun et~al.(1998)LeCun, Bottou, Bengio, and
Haffner]{lecun1998gradient}
Yann LeCun, L{\'e}on Bottou, Yoshua Bengio, and Patrick Haffner.
\newblock Gradient-based learning applied to document recognition.
\newblock \emph{Proceedings of the IEEE}, 86\penalty0 (11):\penalty0
2278--2324, 1998.
\bibitem[LeCun et~al.(2015)LeCun, Bengio, and Hinton]{lecun2015deep}
Yann LeCun, Yoshua Bengio, and Geoffrey Hinton.
\newblock Deep learning.
\newblock \emph{Nature}, 521\penalty0 (7553):\penalty0 436--444, 2015.
\bibitem[Leibniz(1685)]{Leibniz1685}
G.~W. Leibniz.
\newblock \emph{Machina arithmetica in qua non additio tantum et subtractio sed
et multiplicatio nullo, diviso vero paene nullo animi labore peragantur}.
\newblock Hannover, 1685.
\bibitem[Leroy(1997)]{leroy1997effectiveness}
Xavier Leroy.
\newblock The effectiveness of type-based unboxing.
\newblock In \emph{TIC 1997: Workshop Types in Compilation}, 1997.
\bibitem[Linnainmaa(1970)]{linnainmaa1970representation}
Seppo Linnainmaa.
\newblock The representation of the cumulative rounding error of an algorithm
as a taylor expansion of the local rounding errors.
\newblock Master's thesis, University of Helsinki, 1970.
\bibitem[Linnainmaa(1976)]{linnainmaa1976taylor}
Seppo Linnainmaa.
\newblock Taylor expansion of the accumulated rounding error.
\newblock \emph{{BIT} Numerical Mathematics}, 16\penalty0 (2):\penalty0
146--160, 1976.
\bibitem[Loper and Black(2014)]{loper2014opendr}
Matthew~M. Loper and Michael~J. Black.
\newblock {OpenDR}: An approximate differentiable renderer.
\newblock In \emph{European Conference on Computer Vision}, pages 154--169.
Springer, 2014.
\bibitem[Maclaurin(2016)]{maclaurin2016modeling}
Dougal Maclaurin.
\newblock \emph{Modeling, Inference and Optimization with Composable
Differentiable Procedures}.
\newblock PhD thesis, School of Engineering and Applied Sciences, Harvard
University, 2016.
\bibitem[Maclaurin et~al.(2015)Maclaurin, Duvenaud, and Adams]{Maclaurin2015}
Dougal Maclaurin, David Duvenaud, and Ryan Adams.
\newblock Gradient-based hyperparameter optimization through reversible
learning.
\newblock In \emph{International Conference on Machine Learning}, pages
2113--2122, 2015.
\bibitem[Manzyuk et~al.(2012)Manzyuk, Pearlmutter, Radul, Rush, and
Siskind]{manzyuk2012confusion}
Oleksandr Manzyuk, Barak~A. Pearlmutter, Alexey~Andreyevich Radul, David~R
Rush, and Jeffrey~Mark Siskind.
\newblock Confusion of tagged perturbations in forward automatic
differentiation of higher-order functions.
\newblock \emph{arXiv preprint arXiv:1211.4892}, 2012.
\bibitem[Mayne and Jacobson(1970)]{jacobson1970differential}
David~Q. Mayne and David~H. Jacobson.
\newblock \emph{Differential Dynamic Programming}.
\newblock American Elsevier Pub. Co., New York, 1970.
\bibitem[Mazourik(1991)]{Mazourik1991}
Vladimir Mazourik.
\newblock Integration of automatic differentiation into a numerical library for
{PC}'s.
\newblock In A.~Griewank and G.~F. Corliss, editors, \emph{Automatic
Differentiation of Algorithms: Theory, Implementation, and Application},
pages 315--29. Society for Industrial and Applied Mathematics, Philadelphia,
PA, 1991.
\bibitem[Meyer et~al.(2003)Meyer, Fournier, and Berg]{Meyer2003}
Renate Meyer, David~A. Fournier, and Andreas Berg.
\newblock Stochastic volatility: Bayesian computation using automatic
differentiation and the extended {Kalman} filter.
\newblock \emph{Econometrics Journal}, 6\penalty0 (2):\penalty0 408--420, 2003.
\newblock \doi{10.1111/1368-423X.t01-1-00116}.
\bibitem[Michelotti(1990)]{Michelotti1990}
L.~Michelotti.
\newblock {MXYZPTLK}: A practical, user-friendly {C++} implementation of
differential algebra: User's guide.
\newblock Technical Memorandum FN-535, Fermi National Accelerator Laboratory,
Batavia, IL, 1990.
\bibitem[Mikolov et~al.(2010)Mikolov, Karafi{\'a}t, Burget, {\v{C}}ernock{\`y},
and Khudanpur]{mikolov2010recurrent}
Tom{\'a}{\v{s}} Mikolov, Martin Karafi{\'a}t, Luk{\'a}{\v{s}} Burget, Jan
{\v{C}}ernock{\`y}, and Sanjeev Khudanpur.
\newblock Recurrent neural network based language model.
\newblock In \emph{Eleventh Annual Conference of the International Speech
Communication Association}, 2010.
\bibitem[Müller and Cusdin(2005)]{Muller2005}
J.~D. Müller and P.~Cusdin.
\newblock On the performance of discrete adjoint {CFD} codes using automatic
differentiation.
\newblock \emph{International Journal for Numerical Methods in Fluids},
47\penalty0 (8-9):\penalty0 939--945, 2005.
\newblock ISSN 1097-0363.
\newblock \doi{10.1002/fld.885}.
\bibitem[Naumann(2004)]{naumann2004optimal}
Uwe Naumann.
\newblock Optimal accumulation of {Jacobian} matrices by elimination methods on
the dual computational graph.
\newblock \emph{Mathematical Programming}, 99\penalty0 (3):\penalty0 399--421,
2004.
\bibitem[Naumann and Riehme(2005)]{Naumann2005}
Uwe Naumann and Jan Riehme.
\newblock Computing adjoints with the {NAGWare} {F}ortran~95 compiler.
\newblock In H.~M. B{\"u}cker, G.~Corliss, P.~Hovland, U.~Naumann, and
B.~Norris, editors, \emph{Automatic Differentiation: {A}pplications, Theory,
and Implementations}, Lecture Notes in Computational Science and Engineering,
pages 159--69. Springer, 2005.
\bibitem[Neal(1993)]{Neal1993}
Radford~M. Neal.
\newblock Probabilistic inference using {Markov} chain {Monte Carlo} methods.
\newblock Technical Report CRG-TR-93-1, Department of Computer Science,
University of Toronto, 1993.
\bibitem[Neidinger(1989)]{Neidinger1989}
Richard~D. Neidinger.
\newblock Automatic differentiation and {APL}.
\newblock \emph{College Mathematics Journal}, 20\penalty0 (3):\penalty0
238--51, 1989.
\newblock \doi{10.2307/2686776}.
\bibitem[Nolan(1953)]{Nolan1953}
John~F. Nolan.
\newblock Analytical differentiation on a digital computer.
\newblock Master's thesis, Massachusetts Institute of Technology, 1953.
\bibitem[Ostiguy and Michelotti(2007)]{Ostiguy2007}
J.~F. Ostiguy and L.~Michelotti.
\newblock Mxyzptlk: An efficient, native {C++} differentiation engine.
\newblock In \emph{Particle Accelerator Conference (PAC 2007)}, pages 3489--91.
IEEE, 2007.
\newblock \doi{10.1109/PAC.2007.4440468}.
\bibitem[Parker(1985)]{Parker1985}
David~B. Parker.
\newblock Learning-logic: Casting the cortex of the human brain in silicon.
\newblock Technical Report TR-47, Center for Computational Research in
Economics and Management Science, MIT, 1985.
\bibitem[Pascual and Hascoët(2008)]{Pascual2008}
Valérie Pascual and Laurent Hascoët.
\newblock {TAPENADE} for {C}.
\newblock In \emph{Advances in Automatic Differentiation}, Lecture Notes in
Computational Science and Engineering, pages 199--210. Springer, 2008.
\newblock \doi{10.1007/978-3-540-68942-3_18}.
\bibitem[Paszke et~al.(2017)Paszke, Gross, Chintala, Chanan, Yang, DeVito, Lin,
Desmaison, Antiga, and Lerer]{paszke2017automatic}
Adam Paszke, Sam Gross, Soumith Chintala, Gregory Chanan, Edward Yang, Zachary
DeVito, Zeming Lin, Alban Desmaison, Luca Antiga, and Adam Lerer.
\newblock Automatic differentiation in {PyTorch}.
\newblock In \emph{NIPS 2017 Autodiff Workshop: The Future of Gradient-based
Machine Learning Software and Techniques, Long Beach, CA, US, December 9,
2017}, 2017.
\bibitem[Pearlmutter(1994)]{Pearlmutter1994}
Barak~A. Pearlmutter.
\newblock Fast exact multiplication by the {Hessian}.
\newblock \emph{Neural Computation}, 6:\penalty0 147--60, 1994.
\newblock \doi{10.1162/neco.1994.6.1.147}.
\bibitem[Pearlmutter and Siskind(2008)]{pearlmutter2008reverse}
Barak~A. Pearlmutter and Jeffrey~Mark Siskind.
\newblock Reverse-mode {AD} in a functional framework: Lambda the ultimate
backpropagator.
\newblock \emph{ACM Transactions on Programming Languages and Systems
(TOPLAS)}, 30\penalty0 (2):\penalty0 1--36, March 2008.
\newblock \doi{10.1145/1330017.1330018}.
\bibitem[Peng and Robinson(1976)]{Peng1976}
Ding-Yu Peng and Donald~B. Robinson.
\newblock A new two-constant equation of state.
\newblock \emph{Industrial and Engineering Chemistry Fundamentals}, 15\penalty0
(1):\penalty0 59--64, 1976.
\newblock \doi{10.1021/i160057a011}.
\bibitem[Peterson(1989)]{peterson1989untagged}
John Peterson.
\newblock Untagged data in tagged environments: Choosing optimal
representations at compile time.
\newblock In \emph{Proceedings of the Fourth International Conference on
Functional Programming Languages and Computer Architecture}, pages 89--99.
ACM, 1989.
\bibitem[Pfeiffer(1987)]{Pfeiffer1987}
F.~W. Pfeiffer.
\newblock Automatic differentiation in {PROSE}.
\newblock \emph{SIGNUM Newsletter}, 22\penalty0 (1):\penalty0 2--8, 1987.
\newblock \doi{10.1145/24680.24681}.
\bibitem[Pock et~al.(2007)Pock, Pock, and Bischof]{Pock2007}
Thomas Pock, Michael Pock, and Horst Bischof.
\newblock Algorithmic differentiation: Application to variational problems in
computer vision.
\newblock \emph{IEEE Transactions on Pattern Analysis and Machine
Intelligence}, 29\penalty0 (7):\penalty0 1180--1193, 2007.
\newblock \doi{10.1109/TPAMI.2007.1044}.
\bibitem[Press et~al.(2007)Press, Teukolsky, Vetterling, and
Flannery]{Press2007}
William~H. Press, Saul~A. Teukolsky, William~T. Vetterling, and Brian~P.
Flannery.
\newblock \emph{Numerical Recipes: The Art of Scientific Computing}.
\newblock Cambridge University Press, 2007.
\bibitem[Rall(2006)]{Rall2006}
Louise~B. Rall.
\newblock Perspectives on automatic differentiation: Past, present, and future?
\newblock In M.~B\"{u}cker, G.~Corliss, U.~Naumann, P.~Hovland, and B.~Norris,
editors, \emph{Automatic Differentiation: Applications, Theory, and
Implementations}, volume~50 of \emph{Lecture Notes in Computational Science
and Engineering}, pages 1--14. Springer Berlin Heidelberg, 2006.
\bibitem[Rasmussen and Williams(2006)]{Rasmussen2006}
Carl~Edward Rasmussen and Christopher K.~I. Williams.
\newblock \emph{Gaussian processes for machine learning}.
\newblock {MIT} Press, 2006.
\bibitem[Revels et~al.(2016{\natexlab{a}})Revels, Lubin, and
Papamarkou]{RevelsLubinPapamarkou2016}
J.~Revels, M.~Lubin, and T.~Papamarkou.
\newblock Forward-mode automatic differentiation in {Julia}.
\newblock \emph{arXiv:1607.07892 [cs.MS]}, 2016{\natexlab{a}}.
\newblock URL \url{https://arxiv.org/abs/1607.07892}.
\bibitem[Revels et~al.(2016{\natexlab{b}})Revels, Lubin, and
Papamarkou]{revels2016forward}
Jarrett Revels, Miles Lubin, and Theodore Papamarkou.
\newblock Forward-mode automatic differentiation in {J}ulia.
\newblock \emph{arXiv preprint arXiv:1607.07892}, 2016{\natexlab{b}}.
\bibitem[Rezende et~al.(2014)Rezende, Mohamed, and
Wierstra]{rezende2014stochastic}
Danilo~Jimenez Rezende, Shakir Mohamed, and Daan Wierstra.
\newblock Stochastic backpropagation and approximate inference in deep
generative models.
\newblock In \emph{International Conference on Machine Learning}, pages
1278--1286, 2014.
\bibitem[Rich and Hill(1992)]{Hill1992}
Lawrence~C. Rich and David~R. Hill.
\newblock Automatic differentiation in {MATLAB}.
\newblock \emph{Applied Numerical Mathematics}, 9:\penalty0 33--43, 1992.
\bibitem[Ritchie et~al.(2016)Ritchie, Horsfall, and Goodman]{ritchie2016deep}
Daniel Ritchie, Paul Horsfall, and Noah~D Goodman.
\newblock Deep amortized inference for probabilistic programs.
\newblock \emph{arXiv preprint arXiv:1610.05735}, 2016.
\bibitem[Rollins(2009)]{Rollins2009}
Elizabeth Rollins.
\newblock Optimization of neural network feedback control systems using
automatic differentiation.
\newblock Master's thesis, Department of Aeronautics and Astronautics,
Massachusetts Institute of Technology, 2009.
\bibitem[Rozonoer(1959)]{Rozonoer-Pontryagin-1959a}
L.~I. Rozonoer.
\newblock {L}. {S}. {Pontryagin}'s maximum principle in the theory of optimum
systems---{Part} {II}.
\newblock \emph{Automat. i Telemekh.}, 20:\penalty0 1441--1458, 1959.
\bibitem[Rumelhart et~al.(1986)Rumelhart, Hinton, and
Williams]{rumelhart1986learning}
David~E. Rumelhart, Geoffrey~E. Hinton, and Ronald~J. Williams.
\newblock Learning representations by back-propagating errors.
\newblock \emph{Nature}, 323\penalty0 (6088):\penalty0 533, 1986.
\bibitem[Rump(1999)]{Rump1999}
Siegfried~M. Rump.
\newblock {INTLAB}---{INTerval} {LABoratory}.
\newblock In \emph{Developments in Reliable Computing}, pages 77--104. Kluwer
Academic Publishers, Dordrecht, 1999.
\newblock \doi{10.1007/978-94-017-1247-7_7}.
\bibitem[Salimans et~al.(2015)Salimans, Kingma, and Welling]{Salimans2014}
Tim Salimans, Diederik Kingma, and Max Welling.
\newblock Markov chain {Monte Carlo} and variational inference: Bridging the
gap.
\newblock In \emph{Proceedings of the 32nd International Conference on Machine
Learning (ICML-15)}, pages 1218--1226, 2015.
\bibitem[Salvatier et~al.(2016)Salvatier, Wiecki, and
Fonnesbeck]{salvatier2016probabilistic}
John Salvatier, Thomas~V Wiecki, and Christopher Fonnesbeck.
\newblock Probabilistic programming in {Python} using {PyMC3}.
\newblock \emph{PeerJ Computer Science}, 2:\penalty0 e55, 2016.
\bibitem[Schaul et~al.(2013)Schaul, Zhang, and LeCun]{schaul2013no}
Tom Schaul, Sixin Zhang, and Yann LeCun.
\newblock No more pesky learning rates.
\newblock In \emph{International Conference on Machine Learning}, pages
343--351, 2013.
\bibitem[Schmidhuber(2015)]{schmidhuber2015deep}
J{\"u}rgen Schmidhuber.
\newblock Deep learning in neural networks: An overview.
\newblock \emph{Neural Networks}, 61:\penalty0 85--117, 2015.
\bibitem[Schraudolph(1999)]{Schraudolph1999}
Nicol~N. Schraudolph.
\newblock Local gain adaptation in stochastic gradient descent.
\newblock In \emph{Proceedings of the International Conference on Artificial
Neural Networks}, pages 569--74, Edinburgh, Scotland, 1999. IEE London.
\newblock \doi{10.1049/cp:19991170}.
\bibitem[Schraudolph and Graepel(2003)]{Schraudolph2003}
Nicol~N. Schraudolph and Thore Graepel.
\newblock Combining conjugate direction methods with stochastic approximation
of gradients.
\newblock In \emph{Proceedings of the Ninth International Workshop on
Artificial Intelligence and Statistics}, 2003.
\bibitem[Seide and Agarwal(2016)]{seide2016cntk}
Frank Seide and Amit Agarwal.
\newblock {CNTK}: Microsoft's open-source deep-learning toolkit.
\newblock In \emph{Proceedings of the 22Nd ACM SIGKDD International Conference
on Knowledge Discovery and Data Mining}, KDD '16, pages 2135--2135, New York,
NY, USA, 2016. ACM.
\newblock ISBN 978-1-4503-4232-2.
\newblock \doi{10.1145/2939672.2945397}.
\bibitem[Shazeer et~al.(2017)Shazeer, Mirhoseini, Maziarz, Davis, Le, Hinton,
and Dean]{shazeer2017outrageously}
Noam Shazeer, Azalia Mirhoseini, Krzysztof Maziarz, Andy Davis, Quoc Le,
Geoffrey Hinton, and Jeff Dean.
\newblock Outrageously large neural networks: The sparsely-gated
mixture-of-experts layer.
\newblock In \emph{International Conference on Learning Representations 2017},
2017.
\bibitem[Shivers(1991)]{shivers1991control}
Olin Shivers.
\newblock \emph{Control-flow analysis of higher-order languages}.
\newblock PhD thesis, Carnegie Mellon University, 1991.
\bibitem[Shtof et~al.(2013)Shtof, Agathos, Gingold, Shamir, and
{CohenOr}]{Shtof2013}
Alex Shtof, Alexander Agathos, Yotam Gingold, Ariel Shamir, and Daniel
{CohenOr}.
\newblock Geosemantic snapping for sketch-based modeling.
\newblock \emph{Computer Graphics Forum}, 32\penalty0 (2):\penalty0 245--53,
2013.
\newblock \doi{10.1111/cgf.12044}.
\bibitem[Siddharth et~al.(2017)Siddharth, Paige, van~de Meent, Desmaison,
Goodman, Kohli, Wood, and Torr]{siddharth2017learning}
N.~Siddharth, Brooks Paige, Jan-Willem van~de Meent, Alban Desmaison, Noah~D.
Goodman, Pushmeet Kohli, Frank Wood, and Philip Torr.
\newblock Learning disentangled representations with semi-supervised deep
generative models.
\newblock In I.~Guyon, U.~V. Luxburg, S.~Bengio, H.~Wallach, R.~Fergus,
S.~Vishwanathan, and R.~Garnett, editors, \emph{Advances in Neural
Information Processing Systems 30}, pages 5927--5937. Curran Associates,
Inc., 2017.
\bibitem[Simard et~al.(1998)Simard, {LeCun}, Denker, and Victorri]{Simard1998}
Patrice Simard, Yann {LeCun}, John Denker, and Bernard Victorri.
\newblock Transformation invariance in pattern recognition, tangent distance
and tangent propagation.
\newblock In G.~Orr and K.~Muller, editors, \emph{Neural Networks: Tricks of
the Trade}. Springer, 1998.
\bibitem[Sirkes and Tziperman(1997)]{sirkes-tziperman-1997a}
Z.~Sirkes and E.~Tziperman.
\newblock Finite difference of adjoint or adjoint of finite difference?
\newblock \emph{Monthly Weather Review}, 125\penalty0 (12):\penalty0 3373--8,
1997.
\newblock \doi{10.1175/1520-0493(1997)125<3373:FDOAOA>2.0.CO;2}.
\bibitem[Siskind and Pearlmutter(2005)]{SiskindPearlmutter2005a}
Jeffrey~Mark Siskind and Barak~A. Pearlmutter.
\newblock Perturbation confusion and referential transparency: Correct
functional implementation of forward-mode {AD}.
\newblock In Andrew Butterfield, editor, \emph{Implementation and Application
of Functional Languages---17th International Workshop, IFL'05}, pages 1--9,
Dublin, Ireland, 2005.
\newblock Trinity College Dublin Computer Science Department Technical Report
TCD-CS-2005-60.
\bibitem[Siskind and Pearlmutter(2008{\natexlab{a}})]{Siskind2008}
Jeffrey~Mark Siskind and Barak~A. Pearlmutter.
\newblock Using polyvariant union-free flow analysis to compile a higher-order
functional-programming language with a first-class derivative operator to
efficient {Fortran}-like code.
\newblock Technical Report TR-ECE-08-01, School of Electrical and Computer
Engineering, Purdue University, 2008{\natexlab{a}}.
\bibitem[Siskind and Pearlmutter(2008{\natexlab{b}})]{Siskind2008b}
Jeffrey~Mark Siskind and Barak~A. Pearlmutter.
\newblock Nesting forward-mode {AD} in a functional framework.
\newblock \emph{Higher-Order and Symbolic Computation}, 21\penalty0
(4):\penalty0 361--376, 2008{\natexlab{b}}.
\bibitem[Siskind and Pearlmutter(2016)]{siskind2016efficient}
Jeffrey~Mark Siskind and Barak~A. Pearlmutter.
\newblock Efficient implementation of a higher-order language with built-in
{AD}.
\newblock In \emph{7th International Conference on Algorithmic Differentiation,
Christ Church Oxford, UK, September 12--15, 2016}, 2016.
\newblock Also arXiv:1611.03416.
\bibitem[Siskind and Pearlmutter(2017)]{siskind2017divide}
Jeffrey~Mark Siskind and Barak~A. Pearlmutter.
\newblock Divide-and-conquer checkpointing for arbitrary programs with no user
annotation.
\newblock In \emph{NIPS 2017 Autodiff Workshop: The Future of Gradient-based
Machine Learning Software and Techniques, Long Beach, CA, US, December 9,
2017}, 2017.
\newblock Also arXiv:1708.06799.
\bibitem[Slusanschi and Dumitrel(2016)]{slusanschi2016adijac}
Emil~I. Slusanschi and Vlad Dumitrel.
\newblock {ADiJaC}---{A}utomatic differentiation of {J}ava classfiles.
\newblock \emph{ACM Transaction on Mathematical Software}, 43\penalty0
(2):\penalty0 9:1--9:33, September 2016.
\newblock ISSN 0098-3500.
\newblock \doi{10.1145/2904901}.
\bibitem[Speelpenning(1980)]{Speelpenning80}
Bert Speelpenning.
\newblock \emph{Compiling Fast Partial Derivatives of Functions Given by
Algorithms}.
\newblock PhD thesis, Department of Computer Science, University of Illinois at
Urbana-Champaign, 1980.
\bibitem[Sra et~al.(2011)Sra, Nowozin, and Wright]{Sra2011}
Suvrit Sra, Sebastian Nowozin, and Stephen~J. Wright.
\newblock \emph{Optimization for Machine Learning}.
\newblock {MIT} Press, 2011.
\bibitem[Srajer et~al.(2016)Srajer, Kukelova, and
Fitzgibbon]{srajer2016benchmark}
Filip Srajer, Zuzana Kukelova, and Andrew Fitzgibbon.
\newblock A benchmark of selected algorithmic differentiation tools on some
problems in machine learning and computer vision.
\newblock In \emph{AD2016: The 7th International Conference on Algorithmic
Differentiation, Monday 12th--Thursday 15th September 2016, Christ Church
Oxford, UK: Programme and Abstracts}, pages 181--184. Society for Industrial
and Applied Mathematics (SIAM), 2016.
\bibitem[Srinivasan and Todorov(2015)]{srinivasan-todorov-2015a}
Akshay Srinivasan and Emanuel Todorov.
\newblock Graphical {Newton}.
\newblock Technical Report arXiv:1508.00952, arXiv preprint, 2015.
\bibitem[Stuhlm{\"u}ller et~al.(2013)Stuhlm{\"u}ller, Taylor, and
Goodman]{stuhlmuller2013learning}
Andreas Stuhlm{\"u}ller, Jacob Taylor, and Noah Goodman.
\newblock Learning stochastic inverses.
\newblock In \emph{Advances in Neural Information Processing Systems}, pages
3048--3056, 2013.
\bibitem[Such et~al.(2017)Such, Madhavan, Conti, Lehman, Stanley, and
Clune]{such2017deep}
Felipe~Petroski Such, Vashisht Madhavan, Edoardo Conti, Joel Lehman, Kenneth~O.
Stanley, and Jeff Clune.
\newblock Deep neuroevolution: Genetic algorithms are a competitive alternative
for training deep neural networks for reinforcement learning.
\newblock \emph{arXiv preprint arXiv:1712.06567}, 2017.
\bibitem[Sukhbaatar et~al.(2015)Sukhbaatar, Weston, Fergus,
et~al.]{sukhbaatar2015end}
Sainbayar Sukhbaatar, Jason Weston, Rob Fergus, et~al.
\newblock End-to-end memory networks.
\newblock In \emph{Advances in Neural Information Processing Systems}, pages
2440--2448, 2015.
\bibitem[Sussman and Wisdom(2001)]{Sussman2001}
Gerald~J. Sussman and Jack Wisdom.
\newblock \emph{Structure and Interpretation of Classical Mechanics}.
\newblock {MIT} Press, 2001.
\newblock \doi{10.1063/1.1457268}.
\bibitem[Taylor et~al.(2014)Taylor, Stebbing, Ramakrishna, Keskin, Shotton,
Izadi, Hertzmann, and Fitzgibbon]{taylor2014user}
Jonathan Taylor, Richard Stebbing, Varun Ramakrishna, Cem Keskin, Jamie
Shotton, Shahram Izadi, Aaron Hertzmann, and Andrew Fitzgibbon.
\newblock User-specific hand modeling from monocular depth sequences.
\newblock In \emph{Proceedings of the {IEEE} Conference on Computer Vision and
Pattern Recognition}, pages 644--651, 2014.
\bibitem[Thomas et~al.(2006)Thomas, Dowell, and Hall]{thomas2006using}
Jeffrey~P. Thomas, Earl~H. Dowell, and Kenneth~C. Hall.
\newblock Using automatic differentiation to create a nonlinear reduced order
model of a computational fluid dynamic solver.
\newblock \emph{AIAA Paper}, 7115:\penalty0 2006, 2006.
\bibitem[Tieleman and Hinton(2012)]{tieleman2012lecture}
T.~Tieleman and G.~Hinton.
\newblock Lecture 6.5---{RMSProp}: Divide the gradient by a running average of
its recent magnitude.
\newblock \emph{COURSERA: Neural Networks for Machine Learning}, 4\penalty0
(2), 2012.
\bibitem[Tokui et~al.(2015)Tokui, Oono, Hido, and Clayton]{tokui2015chainer}
Seiya Tokui, Kenta Oono, Shohei Hido, and Justin Clayton.
\newblock Chainer: a next-generation open source framework for deep learning.
\newblock In \emph{Proceedings of Workshop on Machine Learning Systems
(LearningSys) in The Twenty-ninth Annual Conference on Neural Information
Processing Systems (NIPS)}, 2015.
\bibitem[Tran et~al.(2016)Tran, Kucukelbir, Dieng, Rudolph, Liang, and
Blei]{tran2016edward}
Dustin Tran, Alp Kucukelbir, Adji~B. Dieng, Maja Rudolph, Dawen Liang, and
David~M. Blei.
\newblock {Edward: A library for probabilistic modeling, inference, and
criticism}.
\newblock \emph{arXiv preprint arXiv:1610.09787}, 2016.
\bibitem[Tran et~al.(2017)Tran, Hoffman, Saurous, Brevdo, Murphy, and
Blei]{tran2017deep}
Dustin Tran, Matthew~D. Hoffman, Rif~A. Saurous, Eugene Brevdo, Kevin Murphy,
and David~M. Blei.
\newblock Deep probabilistic programming.
\newblock In \emph{International Conference on Learning Representations}, 2017.
\bibitem[Triggs et~al.(1999)Triggs, McLauchlan, Hartley, and
Fitzgibbon]{triggs1999bundle}
Bill Triggs, Philip~F. McLauchlan, Richard~I. Hartley, and Andrew~W.
Fitzgibbon.
\newblock Bundle adjustment—a modern synthesis.
\newblock In \emph{International Workshop on Vision Algorithms}, pages
298--372. Springer, 1999.
\bibitem[Tucker et~al.(2017)Tucker, Mnih, Maddison, Lawson, and
Sohl-Dickstein]{tucker2017rebar}
George Tucker, Andriy Mnih, Chris~J. Maddison, John Lawson, and Jascha
Sohl-Dickstein.
\newblock {REBAR}: Low-variance, unbiased gradient estimates for discrete
latent variable models.
\newblock In \emph{Advances in Neural Information Processing Systems}, pages
2624--2633, 2017.
\bibitem[{van Merri{\"e}nboer} et~al.(2017){van Merri{\"e}nboer}, Wiltschko,
and Moldovan]{van2017tangent}
Bart {van Merri{\"e}nboer}, Alexander~B. Wiltschko, and Dan Moldovan.
\newblock Tangent: Automatic differentiation using source code transformation
in {Python}.
\newblock \emph{arXiv preprint arXiv:1711.02712}, 2017.
\bibitem[Verma(2000)]{Verma2000}
Arun Verma.
\newblock An introduction to automatic differentiation.
\newblock \emph{Current Science}, 78\penalty0 (7):\penalty0 804--7, 2000.
\bibitem[Vishwanathan et~al.(2006)Vishwanathan, Schraudolph, Schmidt, and
Murphy]{Vishwanathan2006}
S.~V.~N. Vishwanathan, Nicol~N. Schraudolph, Mark~W. Schmidt, and Kevin~P.
Murphy.
\newblock Accelerated training of conditional random fields with stochastic
gradient methods.
\newblock In \emph{Proceedings of the 23rd International Conference on Machine
Learning (ICML '06)}, pages 969--76, 2006.
\newblock \doi{10.1145/1143844.1143966}.
\bibitem[Walther(2007)]{Walther2007}
Andrea Walther.
\newblock Automatic differentiation of explicit {Runge}-{Kutta} methods for
optimal control.
\newblock \emph{Computational Optimization and Applications}, 36\penalty0
(1):\penalty0 83--108, 2007.
\newblock \doi{10.1007/s10589-006-0397-3}.
\bibitem[Walther and Griewank(2012)]{Walther2012}
Andrea Walther and Andreas Griewank.
\newblock Getting started with {ADOL-C}.
\newblock In U.~Naumann and O.~Schenk, editors, \emph{Combinatorial Scientific
Computing}, chapter~7, pages 181--202. Chapman-Hall CRC Computational
Science, 2012.
\newblock \doi{10.1201/b11644-8}.
\bibitem[Wengert(1964)]{Wengert1964}
Robert~E. Wengert.
\newblock A simple automatic derivative evaluation program.
\newblock \emph{Communications of the {ACM}}, 7:\penalty0 463--4, 1964.
\bibitem[Werbos(1974)]{Werbos-1974a}
Paul~J. Werbos.
\newblock \emph{Beyond Regression: New Tools for Prediction and Analysis in the
Behavioral Sciences}.
\newblock PhD thesis, Harvard University, 1974.
\bibitem[Williams(1992)]{williams1992simple}
Ronald~J. Williams.
\newblock Simple statistical gradient-following algorithms for connectionist
reinforcement learning.
\newblock \emph{Machine Learning}, 8\penalty0 (3-4):\penalty0 229--256, 1992.
\bibitem[Willkomm and Vehreschild(2013)]{Willkomm2013}
J.~Willkomm and A.~Vehreschild.
\newblock The {ADiMat} handbook, 2013.
\newblock URL \url{http://adimat.sc.informatik.tu-darmstadt.de/doc/}.
\bibitem[Wingate et~al.(2011)Wingate, Goodman, Stuhlmüller, and
Siskind]{Wingate2011}
David Wingate, Noah Goodman, Andreas Stuhlmüller, and Jeffrey~Mark Siskind.
\newblock Nonstandard interpretations of probabilistic programs for efficient
inference.
\newblock \emph{Advances in Neural Information Processing Systems}, 23, 2011.
\bibitem[Yang et~al.(2008)Yang, Zhao, Yan, and Chen]{Yang2008}
Weiwei Yang, Yong Zhao, Li~Yan, and Xiaoqian Chen.
\newblock Application of {PID} controller based on {BP} neural network using
automatic differentiation method.
\newblock In F.~Sun, J.~Zhang, Y.~Tan, J.~Cao, and W.~Yu, editors,
\emph{Advances in Neural Networks---ISNN 2008}, volume 5264 of \emph{Lecture
Notes in Computer Science}, pages 702--711. Springer Berlin Heidelberg, 2008.
\newblock \doi{10.1007/978-3-540-87734-9_80}.
\bibitem[Yildirim et~al.(2015)Yildirim, Kulkarni, Freiwald, and
Tenenbaum]{yildirim2015efficient}
Ilker Yildirim, Tejas~D. Kulkarni, Winrich~A. Freiwald, and Joshua~B.
Tenenbaum.
\newblock Efficient and robust analysis-by-synthesis in vision: A computational
framework, behavioral tests, and modeling neuronal representations.
\newblock In \emph{Annual Conference of the Cognitive Science Society}, 2015.
\bibitem[Yu and Siskind(2013)]{Yu2013}
Haonan Yu and Jeffrey~Mark Siskind.
\newblock Grounded language learning from video described with sentences.
\newblock In \emph{Proceedings of the 51st Annual Meeting of the Association
for Computational Linguistics}, pages 53--63, Sofia, Bulgaria, 2013.
Association for Computational Linguistics.
\bibitem[Zaremba et~al.(2016)Zaremba, Mikolov, Joulin, and
Fergus]{zaremba2016learning}
Wojciech Zaremba, Tomas Mikolov, Armand Joulin, and Rob Fergus.
\newblock Learning simple algorithms from examples.
\newblock In \emph{International Conference on Machine Learning}, pages
421--429, 2016.
\bibitem[Zhu et~al.(1997)Zhu, Byrd, Lu, and Nocedal]{Zhu1997}
Ciyou Zhu, Richard~H. Byrd, Peihuang Lu, and Jorge Nocedal.
\newblock Algorithm 778: {L-BFGS-B}: {Fortran} subroutines for large-scale
bound-constrained optimization.
\newblock \emph{ACM Transactions on Mathematical Software (TOMS)}, 23\penalty0
(4):\penalty0 550--60, 1997.
\newblock \doi{10.1145/279232.279236}.
\end{thebibliography}