To add Stochastic Gradient Descent to Documentation (#63805)

author Ilqar Ramazanli <iramazanli@fb.com>

Wed, 8 Sep 2021 22:20:52 +0000 (15:20 -0700)

committer Facebook GitHub Bot <facebook-github-bot@users.noreply.github.com>

Wed, 8 Sep 2021 22:22:30 +0000 (15:22 -0700)
author Ilqar Ramazanli <iramazanli@fb.com>
Wed, 8 Sep 2021 22:20:52 +0000 (15:20 -0700)
committer Facebook GitHub Bot <facebook-github-bot@users.noreply.github.com>
Wed, 8 Sep 2021 22:22:30 +0000 (15:22 -0700)
diff --git a/torch/optim/sgd.py b/torch/optim/sgd.py

index a7a67ff..a4a7222 100644 (file)
--- a/torch/optim/sgd.py
+++ b/torch/optim/sgd.py
@@ -6,6 +6,32 @@ from .optimizer import Optimizer, required
  class SGD(Optimizer):
      r"""Implements stochastic gradient descent (optionally with momentum).
  
+    .. math::
+       \begin{aligned}
+            &\rule{110mm}{0.4pt}                                                                 \\
+            &\textbf{input}      : \gamma \text{ (lr)}, \: \theta_0 \text{ (params)}, \: f(\theta)
+                \text{ (objective)}, \: \lambda \text{ (weight decay)},                          \\
+            &\hspace{13mm} \:\mu \text{ (momentum)}, \:\tau \text{ (dampening)},\:nesterov\\[-1.ex]
+            &\rule{110mm}{0.4pt}                                                                 \\
+            &\textbf{for} \: t=1 \: \textbf{to} \: \ldots \: \textbf{do}                         \\
+            &\hspace{5mm}g_t           \leftarrow   \nabla_{\theta} f_t (\theta_{t-1})           \\
+            &\hspace{5mm}\textbf{if} \: \lambda \neq 0                                           \\
+            &\hspace{10mm} g_t \leftarrow g_t + \lambda  \theta_{t-1}                            \\
+            &\hspace{5mm}\textbf{if} \: \mu \neq 0                                               \\
+            &\hspace{10mm}\textbf{if} \: t > 1                                                   \\
+            &\hspace{15mm} \textbf{b}_t \leftarrow \mu \textbf{b}_{t-1} + (1-\tau) g_t           \\
+            &\hspace{10mm}\textbf{else}                                                          \\
+            &\hspace{15mm} \textbf{b}_t \leftarrow g_t                                           \\
+            &\hspace{10mm}\textbf{if} \: nesterov                                                \\
+            &\hspace{15mm} g_t \leftarrow g_{t-1} + \mu \textbf{b}_t                             \\
+            &\hspace{10mm}\textbf{else}                                                   \\[-1.ex]
+            &\hspace{15mm} g_t  \leftarrow  \textbf{b}_t                                         \\
+            &\hspace{5mm}\theta_t \leftarrow \theta_{t-1} - \gamma g_t                    \\[-1.ex]
+            &\rule{110mm}{0.4pt}                                                          \\[-1.ex]
+            &\bf{return} \:  \theta_t                                                     \\[-1.ex]
+            &\rule{110mm}{0.4pt}                                                          \\[-1.ex]
+       \end{aligned}
+
      Nesterov momentum is based on the formula from
      `On the importance of initialization and momentum in deep learning`__.
author	Ilqar Ramazanli <iramazanli@fb.com>
	Wed, 8 Sep 2021 22:20:52 +0000 (15:20 -0700)
committer	Facebook GitHub Bot <facebook-github-bot@users.noreply.github.com>
	Wed, 8 Sep 2021 22:22:30 +0000 (15:22 -0700)