mirror of
https://github.com/labmlai/annotated_deep_learning_paper_implementations.git
synced 2025-08-14 01:13:00 +08:00
unescape *
This commit is contained in:
File diff suppressed because one or more lines are too long
@ -24,6 +24,8 @@
|
||||
<link rel="shortcut icon" href="/icon.png"/>
|
||||
<link rel="stylesheet" href="../../pylit.css">
|
||||
<link rel="canonical" href="https://nn.labml.ai/diffusion/ddpm/experiment.html"/>
|
||||
<link rel="stylesheet" href="https://cdn.jsdelivr.net/npm/katex@0.13.18/dist/katex.min.css" integrity="sha384-zTROYFVGOfTw7JV7KUu8udsvW2fx4lWOsCEDqhBreBwlHI4ioVRtmIvEThzJHGET" crossorigin="anonymous">
|
||||
|
||||
<!-- Global site tag (gtag.js) - Google Analytics -->
|
||||
<script async src="https://www.googletagmanager.com/gtag/js?id=G-4V3HC8HBLH"></script>
|
||||
<script>
|
||||
@ -68,11 +70,10 @@
|
||||
<a href='#section-0'>#</a>
|
||||
</div>
|
||||
<h1><a href="index.html">Denoising Diffusion Probabilistic Models (DDPM)</a> training</h1>
|
||||
<p>This trains a DDPM based model on CelebA HQ dataset. You can find the download instruction in this
|
||||
<a href="https://forums.fast.ai/t/download-celeba-hq-dataset/45873/3">discussion on fast.ai</a>.
|
||||
Save the images inside <a href="#dataset_path"><code>data/celebA</code> folder</a>.</p>
|
||||
<p>The paper had used a exponential moving average of the model with a decay of $0.9999$. We have skipped this for
|
||||
simplicity.</p>
|
||||
<p>This trains a DDPM based model on CelebA HQ dataset. You can find the download instruction in this <a href="https://forums.fast.ai/t/download-celeba-hq-dataset/45873/3">discussion on fast.ai</a>. Save the images inside <a href="#dataset_path"><code>data/celebA</code>
|
||||
folder</a>.</p>
|
||||
<p>The paper had used a exponential moving average of the model with a decay of <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.64444em;vertical-align:0em;"></span><span class="mord">0.9999</span></span></span></span>. We have skipped this for simplicity.</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">18</span><span></span><span class="kn">from</span> <span class="nn">typing</span> <span class="kn">import</span> <span class="n">List</span>
|
||||
@ -95,6 +96,7 @@ simplicity.</p>
|
||||
<a href='#section-1'>#</a>
|
||||
</div>
|
||||
<h2>Configurations</h2>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">32</span><span class="k">class</span> <span class="nc">Configs</span><span class="p">(</span><span class="n">BaseConfigs</span><span class="p">):</span></pre></div>
|
||||
@ -105,9 +107,9 @@ simplicity.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-2'>#</a>
|
||||
</div>
|
||||
<p>Device to train the model on.
|
||||
<a href="https://docs.labml.ai/api/helpers.html#labml_helpers.device.DeviceConfigs"><code>DeviceConfigs</code></a>
|
||||
picks up an available CUDA device or defaults to CPU.</p>
|
||||
<p>Device to train the model on. <a href="https://docs.labml.ai/api/helpers.html#labml_helpers.device.DeviceConfigs"><code>DeviceConfigs</code>
|
||||
</a> picks up an available CUDA device or defaults to CPU. </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">39</span> <span class="n">device</span><span class="p">:</span> <span class="n">torch</span><span class="o">.</span><span class="n">device</span> <span class="o">=</span> <span class="n">DeviceConfigs</span><span class="p">()</span></pre></div>
|
||||
@ -118,7 +120,8 @@ simplicity.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-3'>#</a>
|
||||
</div>
|
||||
<p>U-Net model for $\color{cyan}{\epsilon_\theta}(x_t, t)$</p>
|
||||
<p>U-Net model for <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em;"></span><span class="mord" style="color:cyan;"><span class="mord" style="color:cyan;"><span class="mord mathnormal" style="color:cyan;">ϵ</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.33610799999999996em;"><span style="top:-2.5500000000000003em;margin-left:0em;margin-right:0.05em;"><span class="pstrut" style="height:2.7em;"></span><span class="sizing reset-size6 size3 mtight" style="color:cyan;"><span class="mord mathnormal mtight" style="margin-right:0.02778em;color:cyan;">θ</span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.15em;"><span></span></span></span></span></span></span></span><span class="mopen" style="color:cyan;">(</span><span class="mord" style="color:cyan;"><span class="mord mathnormal" style="color:cyan;">x</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.2805559999999999em;"><span style="top:-2.5500000000000003em;margin-left:0em;margin-right:0.05em;"><span class="pstrut" style="height:2.7em;"></span><span class="sizing reset-size6 size3 mtight" style="color:cyan;"><span class="mord mathnormal mtight" style="color:cyan;">t</span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.15em;"><span></span></span></span></span></span></span><span class="mpunct" style="color:cyan;">,</span><span class="mspace" style="margin-right:0.16666666666666666em;"></span><span class="mord mathnormal" style="color:cyan;">t</span><span class="mclose" style="color:cyan;">)</span></span></span></span> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">42</span> <span class="n">eps_model</span><span class="p">:</span> <span class="n">UNet</span></pre></div>
|
||||
@ -129,7 +132,8 @@ simplicity.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-4'>#</a>
|
||||
</div>
|
||||
<p><a href="index.html">DDPM algorithm</a></p>
|
||||
<p><a href="index.html">DDPM algorithm</a> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">44</span> <span class="n">diffusion</span><span class="p">:</span> <span class="n">DenoiseDiffusion</span></pre></div>
|
||||
@ -140,7 +144,8 @@ simplicity.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-5'>#</a>
|
||||
</div>
|
||||
<p>Number of channels in the image. $3$ for RGB.</p>
|
||||
<p>Number of channels in the image. <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.64444em;vertical-align:0em;"></span><span class="mord">3</span></span></span></span> for RGB. </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">47</span> <span class="n">image_channels</span><span class="p">:</span> <span class="nb">int</span> <span class="o">=</span> <span class="mi">3</span></pre></div>
|
||||
@ -151,7 +156,8 @@ simplicity.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-6'>#</a>
|
||||
</div>
|
||||
<p>Image size</p>
|
||||
<p>Image size </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">49</span> <span class="n">image_size</span><span class="p">:</span> <span class="nb">int</span> <span class="o">=</span> <span class="mi">32</span></pre></div>
|
||||
@ -162,7 +168,8 @@ simplicity.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-7'>#</a>
|
||||
</div>
|
||||
<p>Number of channels in the initial feature map</p>
|
||||
<p>Number of channels in the initial feature map </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">51</span> <span class="n">n_channels</span><span class="p">:</span> <span class="nb">int</span> <span class="o">=</span> <span class="mi">64</span></pre></div>
|
||||
@ -173,8 +180,9 @@ simplicity.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-8'>#</a>
|
||||
</div>
|
||||
<p>The list of channel numbers at each resolution.
|
||||
The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<p>The list of channel numbers at each resolution. The number of channels is <code>channel_multipliers[i] * n_channels</code>
|
||||
</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">54</span> <span class="n">channel_multipliers</span><span class="p">:</span> <span class="n">List</span><span class="p">[</span><span class="nb">int</span><span class="p">]</span> <span class="o">=</span> <span class="p">[</span><span class="mi">1</span><span class="p">,</span> <span class="mi">2</span><span class="p">,</span> <span class="mi">2</span><span class="p">,</span> <span class="mi">4</span><span class="p">]</span></pre></div>
|
||||
@ -185,7 +193,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-9'>#</a>
|
||||
</div>
|
||||
<p>The list of booleans that indicate whether to use attention at each resolution</p>
|
||||
<p>The list of booleans that indicate whether to use attention at each resolution </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">56</span> <span class="n">is_attention</span><span class="p">:</span> <span class="n">List</span><span class="p">[</span><span class="nb">int</span><span class="p">]</span> <span class="o">=</span> <span class="p">[</span><span class="kc">False</span><span class="p">,</span> <span class="kc">False</span><span class="p">,</span> <span class="kc">False</span><span class="p">,</span> <span class="kc">True</span><span class="p">]</span></pre></div>
|
||||
@ -196,7 +205,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-10'>#</a>
|
||||
</div>
|
||||
<p>Number of time steps $T$</p>
|
||||
<p>Number of time steps <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.68333em;vertical-align:0em;"></span><span class="mord mathnormal" style="margin-right:0.13889em;">T</span></span></span></span> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">59</span> <span class="n">n_steps</span><span class="p">:</span> <span class="nb">int</span> <span class="o">=</span> <span class="mi">1_000</span></pre></div>
|
||||
@ -207,7 +217,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-11'>#</a>
|
||||
</div>
|
||||
<p>Batch size</p>
|
||||
<p>Batch size </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">61</span> <span class="n">batch_size</span><span class="p">:</span> <span class="nb">int</span> <span class="o">=</span> <span class="mi">64</span></pre></div>
|
||||
@ -218,7 +229,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-12'>#</a>
|
||||
</div>
|
||||
<p>Number of samples to generate</p>
|
||||
<p>Number of samples to generate </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">63</span> <span class="n">n_samples</span><span class="p">:</span> <span class="nb">int</span> <span class="o">=</span> <span class="mi">16</span></pre></div>
|
||||
@ -229,7 +241,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-13'>#</a>
|
||||
</div>
|
||||
<p>Learning rate</p>
|
||||
<p>Learning rate </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">65</span> <span class="n">learning_rate</span><span class="p">:</span> <span class="nb">float</span> <span class="o">=</span> <span class="mf">2e-5</span></pre></div>
|
||||
@ -240,7 +253,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-14'>#</a>
|
||||
</div>
|
||||
<p>Number of training epochs</p>
|
||||
<p>Number of training epochs </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">68</span> <span class="n">epochs</span><span class="p">:</span> <span class="nb">int</span> <span class="o">=</span> <span class="mi">1_000</span></pre></div>
|
||||
@ -251,7 +265,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-15'>#</a>
|
||||
</div>
|
||||
<p>Dataset</p>
|
||||
<p>Dataset </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">71</span> <span class="n">dataset</span><span class="p">:</span> <span class="n">torch</span><span class="o">.</span><span class="n">utils</span><span class="o">.</span><span class="n">data</span><span class="o">.</span><span class="n">Dataset</span></pre></div>
|
||||
@ -262,7 +277,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-16'>#</a>
|
||||
</div>
|
||||
<p>Dataloader</p>
|
||||
<p>Dataloader </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">73</span> <span class="n">data_loader</span><span class="p">:</span> <span class="n">torch</span><span class="o">.</span><span class="n">utils</span><span class="o">.</span><span class="n">data</span><span class="o">.</span><span class="n">DataLoader</span></pre></div>
|
||||
@ -273,7 +289,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-17'>#</a>
|
||||
</div>
|
||||
<p>Adam optimizer</p>
|
||||
<p>Adam optimizer </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">76</span> <span class="n">optimizer</span><span class="p">:</span> <span class="n">torch</span><span class="o">.</span><span class="n">optim</span><span class="o">.</span><span class="n">Adam</span></pre></div>
|
||||
@ -295,7 +312,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-19'>#</a>
|
||||
</div>
|
||||
<p>Create $\color{cyan}{\epsilon_\theta}(x_t, t)$ model</p>
|
||||
<p>Create <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em;"></span><span class="mord" style="color:cyan;"><span class="mord" style="color:cyan;"><span class="mord mathnormal" style="color:cyan;">ϵ</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.33610799999999996em;"><span style="top:-2.5500000000000003em;margin-left:0em;margin-right:0.05em;"><span class="pstrut" style="height:2.7em;"></span><span class="sizing reset-size6 size3 mtight" style="color:cyan;"><span class="mord mathnormal mtight" style="margin-right:0.02778em;color:cyan;">θ</span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.15em;"><span></span></span></span></span></span></span></span><span class="mopen" style="color:cyan;">(</span><span class="mord" style="color:cyan;"><span class="mord mathnormal" style="color:cyan;">x</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.2805559999999999em;"><span style="top:-2.5500000000000003em;margin-left:0em;margin-right:0.05em;"><span class="pstrut" style="height:2.7em;"></span><span class="sizing reset-size6 size3 mtight" style="color:cyan;"><span class="mord mathnormal mtight" style="color:cyan;">t</span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.15em;"><span></span></span></span></span></span></span><span class="mpunct" style="color:cyan;">,</span><span class="mspace" style="margin-right:0.16666666666666666em;"></span><span class="mord mathnormal" style="color:cyan;">t</span><span class="mclose" style="color:cyan;">)</span></span></span></span> model </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">80</span> <span class="bp">self</span><span class="o">.</span><span class="n">eps_model</span> <span class="o">=</span> <span class="n">UNet</span><span class="p">(</span>
|
||||
@ -311,7 +329,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-20'>#</a>
|
||||
</div>
|
||||
<p>Create <a href="index.html">DDPM class</a></p>
|
||||
<p>Create <a href="index.html">DDPM class</a> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">88</span> <span class="bp">self</span><span class="o">.</span><span class="n">diffusion</span> <span class="o">=</span> <span class="n">DenoiseDiffusion</span><span class="p">(</span>
|
||||
@ -326,7 +345,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-21'>#</a>
|
||||
</div>
|
||||
<p>Create dataloader</p>
|
||||
<p>Create dataloader </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">95</span> <span class="bp">self</span><span class="o">.</span><span class="n">data_loader</span> <span class="o">=</span> <span class="n">torch</span><span class="o">.</span><span class="n">utils</span><span class="o">.</span><span class="n">data</span><span class="o">.</span><span class="n">DataLoader</span><span class="p">(</span><span class="bp">self</span><span class="o">.</span><span class="n">dataset</span><span class="p">,</span> <span class="bp">self</span><span class="o">.</span><span class="n">batch_size</span><span class="p">,</span> <span class="n">shuffle</span><span class="o">=</span><span class="kc">True</span><span class="p">,</span> <span class="n">pin_memory</span><span class="o">=</span><span class="kc">True</span><span class="p">)</span></pre></div>
|
||||
@ -337,7 +357,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-22'>#</a>
|
||||
</div>
|
||||
<p>Create optimizer</p>
|
||||
<p>Create optimizer </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">97</span> <span class="bp">self</span><span class="o">.</span><span class="n">optimizer</span> <span class="o">=</span> <span class="n">torch</span><span class="o">.</span><span class="n">optim</span><span class="o">.</span><span class="n">Adam</span><span class="p">(</span><span class="bp">self</span><span class="o">.</span><span class="n">eps_model</span><span class="o">.</span><span class="n">parameters</span><span class="p">(),</span> <span class="n">lr</span><span class="o">=</span><span class="bp">self</span><span class="o">.</span><span class="n">learning_rate</span><span class="p">)</span></pre></div>
|
||||
@ -348,7 +369,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-23'>#</a>
|
||||
</div>
|
||||
<p>Image logging</p>
|
||||
<p>Image logging </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">100</span> <span class="n">tracker</span><span class="o">.</span><span class="n">set_image</span><span class="p">(</span><span class="s2">"sample"</span><span class="p">,</span> <span class="kc">True</span><span class="p">)</span></pre></div>
|
||||
@ -360,6 +382,7 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<a href='#section-24'>#</a>
|
||||
</div>
|
||||
<h3>Sample images</h3>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">102</span> <span class="k">def</span> <span class="nf">sample</span><span class="p">(</span><span class="bp">self</span><span class="p">):</span></pre></div>
|
||||
@ -381,7 +404,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-26'>#</a>
|
||||
</div>
|
||||
<p>$x_T \sim p(x_T) = \mathcal{N}(x_T; \mathbf{0}, \mathbf{I})$</p>
|
||||
<p><span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.58056em;vertical-align:-0.15em;"></span><span class="mord"><span class="mord mathnormal">x</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.32833099999999993em;"><span style="top:-2.5500000000000003em;margin-left:0em;margin-right:0.05em;"><span class="pstrut" style="height:2.7em;"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mathnormal mtight" style="margin-right:0.13889em;">T</span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.15em;"><span></span></span></span></span></span></span><span class="mspace" style="margin-right:0.2777777777777778em;"></span><span class="mrel">∼</span><span class="mspace" style="margin-right:0.2777777777777778em;"></span></span><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em;"></span><span class="mord mathnormal">p</span><span class="mopen">(</span><span class="mord"><span class="mord mathnormal">x</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.32833099999999993em;"><span style="top:-2.5500000000000003em;margin-left:0em;margin-right:0.05em;"><span class="pstrut" style="height:2.7em;"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mathnormal mtight" style="margin-right:0.13889em;">T</span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.15em;"><span></span></span></span></span></span></span><span class="mclose">)</span><span class="mspace" style="margin-right:0.2777777777777778em;"></span><span class="mrel">=</span><span class="mspace" style="margin-right:0.2777777777777778em;"></span></span><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em;"></span><span class="mord mathcal" style="margin-right:0.14736em;">N</span><span class="mopen">(</span><span class="mord"><span class="mord mathnormal">x</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.32833099999999993em;"><span style="top:-2.5500000000000003em;margin-left:0em;margin-right:0.05em;"><span class="pstrut" style="height:2.7em;"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mathnormal mtight" style="margin-right:0.13889em;">T</span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.15em;"><span></span></span></span></span></span></span><span class="mpunct">;</span><span class="mspace" style="margin-right:0.16666666666666666em;"></span><span class="mord mathbf">0</span><span class="mpunct">,</span><span class="mspace" style="margin-right:0.16666666666666666em;"></span><span class="mord mathbf">I</span><span class="mclose">)</span></span></span></span> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">108</span> <span class="n">x</span> <span class="o">=</span> <span class="n">torch</span><span class="o">.</span><span class="n">randn</span><span class="p">([</span><span class="bp">self</span><span class="o">.</span><span class="n">n_samples</span><span class="p">,</span> <span class="bp">self</span><span class="o">.</span><span class="n">image_channels</span><span class="p">,</span> <span class="bp">self</span><span class="o">.</span><span class="n">image_size</span><span class="p">,</span> <span class="bp">self</span><span class="o">.</span><span class="n">image_size</span><span class="p">],</span>
|
||||
@ -393,7 +417,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-27'>#</a>
|
||||
</div>
|
||||
<p>Remove noise for $T$ steps</p>
|
||||
<p>Remove noise for <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.68333em;vertical-align:0em;"></span><span class="mord mathnormal" style="margin-right:0.13889em;">T</span></span></span></span> steps </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">112</span> <span class="k">for</span> <span class="n">t_</span> <span class="ow">in</span> <span class="n">monit</span><span class="o">.</span><span class="n">iterate</span><span class="p">(</span><span class="s1">'Sample'</span><span class="p">,</span> <span class="bp">self</span><span class="o">.</span><span class="n">n_steps</span><span class="p">):</span></pre></div>
|
||||
@ -404,7 +429,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-28'>#</a>
|
||||
</div>
|
||||
<p>$t$</p>
|
||||
<p><span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.61508em;vertical-align:0em;"></span><span class="mord mathnormal">t</span></span></span></span> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">114</span> <span class="n">t</span> <span class="o">=</span> <span class="bp">self</span><span class="o">.</span><span class="n">n_steps</span> <span class="o">-</span> <span class="n">t_</span> <span class="o">-</span> <span class="mi">1</span></pre></div>
|
||||
@ -415,7 +441,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-29'>#</a>
|
||||
</div>
|
||||
<p>Sample from $\color{cyan}{p_\theta}(x_{t-1}|x_t)$</p>
|
||||
<p>Sample from <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em;"></span><span class="mord" style="color:cyan;"><span class="mord" style="color:cyan;"><span class="mord mathnormal" style="color:cyan;">p</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.33610799999999996em;"><span style="top:-2.5500000000000003em;margin-left:0em;margin-right:0.05em;"><span class="pstrut" style="height:2.7em;"></span><span class="sizing reset-size6 size3 mtight" style="color:cyan;"><span class="mord mathnormal mtight" style="margin-right:0.02778em;color:cyan;">θ</span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.15em;"><span></span></span></span></span></span></span></span><span class="mopen" style="color:cyan;">(</span><span class="mord" style="color:cyan;"><span class="mord mathnormal" style="color:cyan;">x</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.301108em;"><span style="top:-2.5500000000000003em;margin-left:0em;margin-right:0.05em;"><span class="pstrut" style="height:2.7em;"></span><span class="sizing reset-size6 size3 mtight" style="color:cyan;"><span class="mord mtight" style="color:cyan;"><span class="mord mathnormal mtight" style="color:cyan;">t</span><span class="mbin mtight" style="color:cyan;">−</span><span class="mord mtight" style="color:cyan;">1</span></span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.208331em;"><span></span></span></span></span></span></span><span class="mord" style="color:cyan;">∣</span><span class="mord" style="color:cyan;"><span class="mord mathnormal" style="color:cyan;">x</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.2805559999999999em;"><span style="top:-2.5500000000000003em;margin-left:0em;margin-right:0.05em;"><span class="pstrut" style="height:2.7em;"></span><span class="sizing reset-size6 size3 mtight" style="color:cyan;"><span class="mord mathnormal mtight" style="color:cyan;">t</span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.15em;"><span></span></span></span></span></span></span><span class="mclose" style="color:cyan;">)</span></span></span></span> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">116</span> <span class="n">x</span> <span class="o">=</span> <span class="bp">self</span><span class="o">.</span><span class="n">diffusion</span><span class="o">.</span><span class="n">p_sample</span><span class="p">(</span><span class="n">x</span><span class="p">,</span> <span class="n">x</span><span class="o">.</span><span class="n">new_full</span><span class="p">((</span><span class="bp">self</span><span class="o">.</span><span class="n">n_samples</span><span class="p">,),</span> <span class="n">t</span><span class="p">,</span> <span class="n">dtype</span><span class="o">=</span><span class="n">torch</span><span class="o">.</span><span class="n">long</span><span class="p">))</span></pre></div>
|
||||
@ -426,7 +453,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-30'>#</a>
|
||||
</div>
|
||||
<p>Log samples</p>
|
||||
<p>Log samples </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">119</span> <span class="n">tracker</span><span class="o">.</span><span class="n">save</span><span class="p">(</span><span class="s1">'sample'</span><span class="p">,</span> <span class="n">x</span><span class="p">)</span></pre></div>
|
||||
@ -438,6 +466,7 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<a href='#section-31'>#</a>
|
||||
</div>
|
||||
<h3>Train</h3>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">121</span> <span class="k">def</span> <span class="nf">train</span><span class="p">(</span><span class="bp">self</span><span class="p">):</span></pre></div>
|
||||
@ -448,7 +477,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-32'>#</a>
|
||||
</div>
|
||||
<p>Iterate through the dataset</p>
|
||||
<p>Iterate through the dataset </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">127</span> <span class="k">for</span> <span class="n">data</span> <span class="ow">in</span> <span class="n">monit</span><span class="o">.</span><span class="n">iterate</span><span class="p">(</span><span class="s1">'Train'</span><span class="p">,</span> <span class="bp">self</span><span class="o">.</span><span class="n">data_loader</span><span class="p">):</span></pre></div>
|
||||
@ -459,7 +489,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-33'>#</a>
|
||||
</div>
|
||||
<p>Increment global step</p>
|
||||
<p>Increment global step </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">129</span> <span class="n">tracker</span><span class="o">.</span><span class="n">add_global_step</span><span class="p">()</span></pre></div>
|
||||
@ -470,7 +501,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-34'>#</a>
|
||||
</div>
|
||||
<p>Move data to device</p>
|
||||
<p>Move data to device </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">131</span> <span class="n">data</span> <span class="o">=</span> <span class="n">data</span><span class="o">.</span><span class="n">to</span><span class="p">(</span><span class="bp">self</span><span class="o">.</span><span class="n">device</span><span class="p">)</span></pre></div>
|
||||
@ -481,7 +513,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-35'>#</a>
|
||||
</div>
|
||||
<p>Make the gradients zero</p>
|
||||
<p>Make the gradients zero </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">134</span> <span class="bp">self</span><span class="o">.</span><span class="n">optimizer</span><span class="o">.</span><span class="n">zero_grad</span><span class="p">()</span></pre></div>
|
||||
@ -492,7 +525,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-36'>#</a>
|
||||
</div>
|
||||
<p>Calculate loss</p>
|
||||
<p>Calculate loss </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">136</span> <span class="n">loss</span> <span class="o">=</span> <span class="bp">self</span><span class="o">.</span><span class="n">diffusion</span><span class="o">.</span><span class="n">loss</span><span class="p">(</span><span class="n">data</span><span class="p">)</span></pre></div>
|
||||
@ -503,7 +537,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-37'>#</a>
|
||||
</div>
|
||||
<p>Compute gradients</p>
|
||||
<p>Compute gradients </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">138</span> <span class="n">loss</span><span class="o">.</span><span class="n">backward</span><span class="p">()</span></pre></div>
|
||||
@ -514,7 +549,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-38'>#</a>
|
||||
</div>
|
||||
<p>Take an optimization step</p>
|
||||
<p>Take an optimization step </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">140</span> <span class="bp">self</span><span class="o">.</span><span class="n">optimizer</span><span class="o">.</span><span class="n">step</span><span class="p">()</span></pre></div>
|
||||
@ -525,7 +561,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-39'>#</a>
|
||||
</div>
|
||||
<p>Track the loss</p>
|
||||
<p>Track the loss </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">142</span> <span class="n">tracker</span><span class="o">.</span><span class="n">save</span><span class="p">(</span><span class="s1">'loss'</span><span class="p">,</span> <span class="n">loss</span><span class="p">)</span></pre></div>
|
||||
@ -537,6 +574,7 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<a href='#section-40'>#</a>
|
||||
</div>
|
||||
<h3>Training loop</h3>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">144</span> <span class="k">def</span> <span class="nf">run</span><span class="p">(</span><span class="bp">self</span><span class="p">):</span></pre></div>
|
||||
@ -558,7 +596,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-42'>#</a>
|
||||
</div>
|
||||
<p>Train the model</p>
|
||||
<p>Train the model </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">150</span> <span class="bp">self</span><span class="o">.</span><span class="n">train</span><span class="p">()</span></pre></div>
|
||||
@ -569,7 +608,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-43'>#</a>
|
||||
</div>
|
||||
<p>Sample some images</p>
|
||||
<p>Sample some images </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">152</span> <span class="bp">self</span><span class="o">.</span><span class="n">sample</span><span class="p">()</span></pre></div>
|
||||
@ -580,7 +620,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-44'>#</a>
|
||||
</div>
|
||||
<p>New line in the console</p>
|
||||
<p>New line in the console </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">154</span> <span class="n">tracker</span><span class="o">.</span><span class="n">new_line</span><span class="p">()</span></pre></div>
|
||||
@ -591,7 +632,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-45'>#</a>
|
||||
</div>
|
||||
<p>Save the model</p>
|
||||
<p>Save the model </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">156</span> <span class="n">experiment</span><span class="o">.</span><span class="n">save_checkpoint</span><span class="p">()</span></pre></div>
|
||||
@ -603,6 +645,7 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<a href='#section-46'>#</a>
|
||||
</div>
|
||||
<h3>CelebA HQ dataset</h3>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">159</span><span class="k">class</span> <span class="nc">CelebADataset</span><span class="p">(</span><span class="n">torch</span><span class="o">.</span><span class="n">utils</span><span class="o">.</span><span class="n">data</span><span class="o">.</span><span class="n">Dataset</span><span class="p">):</span></pre></div>
|
||||
@ -625,7 +668,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-48'>#</a>
|
||||
</div>
|
||||
<p>CelebA images folder</p>
|
||||
<p>CelebA images folder </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">168</span> <span class="n">folder</span> <span class="o">=</span> <span class="n">lab</span><span class="o">.</span><span class="n">get_data_path</span><span class="p">()</span> <span class="o">/</span> <span class="s1">'celebA'</span></pre></div>
|
||||
@ -636,7 +680,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-49'>#</a>
|
||||
</div>
|
||||
<p>List of files</p>
|
||||
<p>List of files </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">170</span> <span class="bp">self</span><span class="o">.</span><span class="n">_files</span> <span class="o">=</span> <span class="p">[</span><span class="n">p</span> <span class="k">for</span> <span class="n">p</span> <span class="ow">in</span> <span class="n">folder</span><span class="o">.</span><span class="n">glob</span><span class="p">(</span><span class="sa">f</span><span class="s1">'**/*.jpg'</span><span class="p">)]</span></pre></div>
|
||||
@ -647,7 +692,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-50'>#</a>
|
||||
</div>
|
||||
<p>Transformations to resize the image and convert to tensor</p>
|
||||
<p>Transformations to resize the image and convert to tensor </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">173</span> <span class="bp">self</span><span class="o">.</span><span class="n">_transform</span> <span class="o">=</span> <span class="n">torchvision</span><span class="o">.</span><span class="n">transforms</span><span class="o">.</span><span class="n">Compose</span><span class="p">([</span>
|
||||
@ -661,7 +707,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-51'>#</a>
|
||||
</div>
|
||||
<p>Size of the dataset</p>
|
||||
<p> Size of the dataset</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">178</span> <span class="k">def</span> <span class="fm">__len__</span><span class="p">(</span><span class="bp">self</span><span class="p">):</span></pre></div>
|
||||
@ -683,7 +730,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-53'>#</a>
|
||||
</div>
|
||||
<p>Get an image</p>
|
||||
<p> Get an image</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">184</span> <span class="k">def</span> <span class="fm">__getitem__</span><span class="p">(</span><span class="bp">self</span><span class="p">,</span> <span class="n">index</span><span class="p">:</span> <span class="nb">int</span><span class="p">):</span></pre></div>
|
||||
@ -706,7 +754,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-55'>#</a>
|
||||
</div>
|
||||
<p>Create CelebA dataset</p>
|
||||
<p> Create CelebA dataset</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">192</span><span class="nd">@option</span><span class="p">(</span><span class="n">Configs</span><span class="o">.</span><span class="n">dataset</span><span class="p">,</span> <span class="s1">'CelebA'</span><span class="p">)</span>
|
||||
@ -730,6 +779,7 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<a href='#section-57'>#</a>
|
||||
</div>
|
||||
<h3>MNIST dataset</h3>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">200</span><span class="k">class</span> <span class="nc">MNISTDataset</span><span class="p">(</span><span class="n">torchvision</span><span class="o">.</span><span class="n">datasets</span><span class="o">.</span><span class="n">MNIST</span><span class="p">):</span></pre></div>
|
||||
@ -769,7 +819,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-60'>#</a>
|
||||
</div>
|
||||
<p>Create MNIST dataset</p>
|
||||
<p> Create MNIST dataset</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">217</span><span class="nd">@option</span><span class="p">(</span><span class="n">Configs</span><span class="o">.</span><span class="n">dataset</span><span class="p">,</span> <span class="s1">'MNIST'</span><span class="p">)</span>
|
||||
@ -803,7 +854,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-63'>#</a>
|
||||
</div>
|
||||
<p>Create experiment</p>
|
||||
<p>Create experiment </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">227</span> <span class="n">experiment</span><span class="o">.</span><span class="n">create</span><span class="p">(</span><span class="n">name</span><span class="o">=</span><span class="s1">'diffuse'</span><span class="p">)</span></pre></div>
|
||||
@ -814,7 +866,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-64'>#</a>
|
||||
</div>
|
||||
<p>Create configurations</p>
|
||||
<p>Create configurations </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">230</span> <span class="n">configs</span> <span class="o">=</span> <span class="n">Configs</span><span class="p">()</span></pre></div>
|
||||
@ -825,7 +878,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-65'>#</a>
|
||||
</div>
|
||||
<p>Set configurations. You can override the defaults by passing the values in the dictionary.</p>
|
||||
<p>Set configurations. You can override the defaults by passing the values in the dictionary. </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">233</span> <span class="n">experiment</span><span class="o">.</span><span class="n">configs</span><span class="p">(</span><span class="n">configs</span><span class="p">,</span> <span class="p">{</span>
|
||||
@ -837,7 +891,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-66'>#</a>
|
||||
</div>
|
||||
<p>Initialize</p>
|
||||
<p>Initialize </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">237</span> <span class="n">configs</span><span class="o">.</span><span class="n">init</span><span class="p">()</span></pre></div>
|
||||
@ -848,7 +903,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-67'>#</a>
|
||||
</div>
|
||||
<p>Set models for saving and loading</p>
|
||||
<p>Set models for saving and loading </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">240</span> <span class="n">experiment</span><span class="o">.</span><span class="n">add_pytorch_models</span><span class="p">({</span><span class="s1">'eps_model'</span><span class="p">:</span> <span class="n">configs</span><span class="o">.</span><span class="n">eps_model</span><span class="p">})</span></pre></div>
|
||||
@ -859,7 +915,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-68'>#</a>
|
||||
</div>
|
||||
<p>Start and run the training loop</p>
|
||||
<p>Start and run the training loop </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">243</span> <span class="k">with</span> <span class="n">experiment</span><span class="o">.</span><span class="n">start</span><span class="p">():</span>
|
||||
@ -871,7 +928,8 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-69'>#</a>
|
||||
</div>
|
||||
|
||||
<p> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">248</span><span class="k">if</span> <span class="vm">__name__</span> <span class="o">==</span> <span class="s1">'__main__'</span><span class="p">:</span>
|
||||
@ -883,24 +941,6 @@ The number of channels is <code>channel_multipliers[i] * n_channels</code></p>
|
||||
<a href="https://labml.ai">labml.ai</a>
|
||||
</div>
|
||||
</div>
|
||||
<script src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.4/MathJax.js?config=TeX-AMS_HTML">
|
||||
</script>
|
||||
<!-- MathJax configuration -->
|
||||
<script type="text/x-mathjax-config">
|
||||
MathJax.Hub.Config({
|
||||
tex2jax: {
|
||||
inlineMath: [ ['$','$'] ],
|
||||
displayMath: [ ['$$','$$'] ],
|
||||
processEscapes: true,
|
||||
processEnvironments: true
|
||||
},
|
||||
// Center justify equations in code and markdown cells. Elsewhere
|
||||
// we use CSS to left justify single line equations in code cells.
|
||||
displayAlign: 'center',
|
||||
"HTML-CSS": { fonts: ["TeX"] }
|
||||
});
|
||||
|
||||
</script>
|
||||
<script>
|
||||
function handleImages() {
|
||||
var images = document.querySelectorAll('p>img')
|
||||
|
File diff suppressed because one or more lines are too long
@ -24,6 +24,8 @@
|
||||
<link rel="shortcut icon" href="/icon.png"/>
|
||||
<link rel="stylesheet" href="../../pylit.css">
|
||||
<link rel="canonical" href="https://nn.labml.ai/diffusion/ddpm/readme.html"/>
|
||||
<link rel="stylesheet" href="https://cdn.jsdelivr.net/npm/katex@0.13.18/dist/katex.min.css" integrity="sha384-zTROYFVGOfTw7JV7KUu8udsvW2fx4lWOsCEDqhBreBwlHI4ioVRtmIvEThzJHGET" crossorigin="anonymous">
|
||||
|
||||
<!-- Global site tag (gtag.js) - Google Analytics -->
|
||||
<script async src="https://www.googletagmanager.com/gtag/js?id=G-4V3HC8HBLH"></script>
|
||||
<script>
|
||||
@ -68,16 +70,11 @@
|
||||
<a href='#section-0'>#</a>
|
||||
</div>
|
||||
<h1><a href="https://nn.labml.ai/diffusion/ddpm/index.html">Denoising Diffusion Probabilistic Models (DDPM)</a></h1>
|
||||
<p>This is a <a href="https://pytorch.org">PyTorch</a> implementation/tutorial of the paper
|
||||
<a href="https://papers.labml.ai/paper/2006.11239">Denoising Diffusion Probabilistic Models</a>.</p>
|
||||
<p>In simple terms, we get an image from data and add noise step by step.
|
||||
Then We train a model to predict that noise at each step and use the model to
|
||||
generate images.</p>
|
||||
<p>Here is the <a href="https://nn.labml.ai/diffusion/ddpm/unet.html">UNet model</a> that predicts the noise and
|
||||
<a href="https://nn.labml.ai/diffusion/ddpm/experiment.html">training code</a>.
|
||||
<a href="https://nn.labml.ai/diffusion/ddpm/evaluate.html">This file</a> can generate samples and interpolations
|
||||
from a trained model.</p>
|
||||
<p><a href="https://app.labml.ai/run/a44333ea251411ec8007d1a1762ed686"><img alt="View Run" src="https://img.shields.io/badge/labml-experiment-brightgreen" /></a></p>
|
||||
<p>This is a <a href="https://pytorch.org">PyTorch</a> implementation/tutorial of the paper <a href="https://papers.labml.ai/paper/2006.11239">Denoising Diffusion Probabilistic Models</a>.</p>
|
||||
<p>In simple terms, we get an image from data and add noise step by step. Then We train a model to predict that noise at each step and use the model to generate images.</p>
|
||||
<p>Here is the <a href="https://nn.labml.ai/diffusion/ddpm/unet.html">UNet model</a> that predicts the noise and <a href="https://nn.labml.ai/diffusion/ddpm/experiment.html">training code</a>. <a href="https://nn.labml.ai/diffusion/ddpm/evaluate.html">This file</a> can generate samples and interpolations from a trained model.</p>
|
||||
<p><a href="https://app.labml.ai/run/a44333ea251411ec8007d1a1762ed686"><img alt="View Run" src="https://img.shields.io/badge/labml-experiment-brightgreen"></a> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
|
||||
@ -88,24 +85,6 @@ from a trained model.</p>
|
||||
<a href="https://labml.ai">labml.ai</a>
|
||||
</div>
|
||||
</div>
|
||||
<script src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.4/MathJax.js?config=TeX-AMS_HTML">
|
||||
</script>
|
||||
<!-- MathJax configuration -->
|
||||
<script type="text/x-mathjax-config">
|
||||
MathJax.Hub.Config({
|
||||
tex2jax: {
|
||||
inlineMath: [ ['$','$'] ],
|
||||
displayMath: [ ['$$','$$'] ],
|
||||
processEscapes: true,
|
||||
processEnvironments: true
|
||||
},
|
||||
// Center justify equations in code and markdown cells. Elsewhere
|
||||
// we use CSS to left justify single line equations in code cells.
|
||||
displayAlign: 'center',
|
||||
"HTML-CSS": { fonts: ["TeX"] }
|
||||
});
|
||||
|
||||
</script>
|
||||
<script>
|
||||
function handleImages() {
|
||||
var images = document.querySelectorAll('p>img')
|
||||
|
@ -24,6 +24,8 @@
|
||||
<link rel="shortcut icon" href="/icon.png"/>
|
||||
<link rel="stylesheet" href="../../pylit.css">
|
||||
<link rel="canonical" href="https://nn.labml.ai/diffusion/ddpm/unet.html"/>
|
||||
<link rel="stylesheet" href="https://cdn.jsdelivr.net/npm/katex@0.13.18/dist/katex.min.css" integrity="sha384-zTROYFVGOfTw7JV7KUu8udsvW2fx4lWOsCEDqhBreBwlHI4ioVRtmIvEThzJHGET" crossorigin="anonymous">
|
||||
|
||||
<!-- Global site tag (gtag.js) - Google Analytics -->
|
||||
<script async src="https://www.googletagmanager.com/gtag/js?id=G-4V3HC8HBLH"></script>
|
||||
<script>
|
||||
@ -68,15 +70,11 @@
|
||||
<a href='#section-0'>#</a>
|
||||
</div>
|
||||
<h1>U-Net model for <a href="index.html">Denoising Diffusion Probabilistic Models (DDPM)</a></h1>
|
||||
<p>This is a <a href="https://papers.labml.ai/paper/1505.04597">U-Net</a> based model to predict noise
|
||||
$\color{cyan}{\epsilon_\theta}(x_t, t)$.</p>
|
||||
<p>U-Net is a gets it’s name from the U shape in the model diagram.
|
||||
It processes a given image by progressively lowering (halving) the feature map resolution and then
|
||||
increasing the resolution.
|
||||
There are pass-through connection at each resolution.</p>
|
||||
<p><img alt="U-Net diagram from paper" src="unet.png" /></p>
|
||||
<p>This implementation contains a bunch of modifications to original U-Net (residual blocks, multi-head attention)
|
||||
and also adds time-step embeddings $t$.</p>
|
||||
<p>This is a <a href="https://papers.labml.ai/paper/1505.04597">U-Net</a> based model to predict noise <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em;"></span><span class="mord" style="color:cyan;"><span class="mord" style="color:cyan;"><span class="mord mathnormal" style="color:cyan;">ϵ</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.33610799999999996em;"><span style="top:-2.5500000000000003em;margin-left:0em;margin-right:0.05em;"><span class="pstrut" style="height:2.7em;"></span><span class="sizing reset-size6 size3 mtight" style="color:cyan;"><span class="mord mathnormal mtight" style="margin-right:0.02778em;color:cyan;">θ</span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.15em;"><span></span></span></span></span></span></span></span><span class="mopen" style="color:cyan;">(</span><span class="mord" style="color:cyan;"><span class="mord mathnormal" style="color:cyan;">x</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.2805559999999999em;"><span style="top:-2.5500000000000003em;margin-left:0em;margin-right:0.05em;"><span class="pstrut" style="height:2.7em;"></span><span class="sizing reset-size6 size3 mtight" style="color:cyan;"><span class="mord mathnormal mtight" style="color:cyan;">t</span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.15em;"><span></span></span></span></span></span></span><span class="mpunct" style="color:cyan;">,</span><span class="mspace" style="margin-right:0.16666666666666666em;"></span><span class="mord mathnormal" style="color:cyan;">t</span><span class="mclose" style="color:cyan;">)</span></span></span></span>.</p>
|
||||
<p>U-Net is a gets it's name from the U shape in the model diagram. It processes a given image by progressively lowering (halving) the feature map resolution and then increasing the resolution. There are pass-through connection at each resolution.</p>
|
||||
<p><img alt="U-Net diagram from paper" src="unet.png"></p>
|
||||
<p>This implementation contains a bunch of modifications to original U-Net (residual blocks, multi-head attention) and also adds time-step embeddings <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.61508em;vertical-align:0em;"></span><span class="mord mathnormal">t</span></span></span></span>.</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">24</span><span></span><span class="kn">import</span> <span class="nn">math</span>
|
||||
@ -94,9 +92,8 @@ There are pass-through connection at each resolution.</p>
|
||||
<a href='#section-1'>#</a>
|
||||
</div>
|
||||
<h3>Swish actiavation function</h3>
|
||||
<p>
|
||||
<script type="math/tex; mode=display">x \cdot \sigma(x)</script>
|
||||
</p>
|
||||
<p><span class="katex-display"><span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.44445em;vertical-align:0em;"></span><span class="mord mathnormal">x</span><span class="mspace" style="margin-right:0.2222222222222222em;"></span><span class="mbin">⋅</span><span class="mspace" style="margin-right:0.2222222222222222em;"></span></span><span class="base"><span class="strut" style="height:1em;vertical-align:-0.25em;"></span><span class="mord mathnormal" style="margin-right:0.03588em;">σ</span><span class="mopen">(</span><span class="mord mathnormal">x</span><span class="mclose">)</span></span></span></span></span></p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">33</span><span class="k">class</span> <span class="nc">Swish</span><span class="p">(</span><span class="n">Module</span><span class="p">):</span></pre></div>
|
||||
@ -119,7 +116,8 @@ There are pass-through connection at each resolution.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-3'>#</a>
|
||||
</div>
|
||||
<h3>Embeddings for $t$</h3>
|
||||
<h3>Embeddings for <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.61508em;vertical-align:0em;"></span><span class="mord mathnormal">t</span></span></span></span></h3>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">44</span><span class="k">class</span> <span class="nc">TimeEmbedding</span><span class="p">(</span><span class="n">nn</span><span class="o">.</span><span class="n">Module</span><span class="p">):</span></pre></div>
|
||||
@ -130,9 +128,9 @@ There are pass-through connection at each resolution.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-4'>#</a>
|
||||
</div>
|
||||
<ul>
|
||||
<li><code>n_channels</code> is the number of dimensions in the embedding</li>
|
||||
</ul>
|
||||
<ul><li><code>n_channels</code>
|
||||
is the number of dimensions in the embedding</li></ul>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">49</span> <span class="k">def</span> <span class="fm">__init__</span><span class="p">(</span><span class="bp">self</span><span class="p">,</span> <span class="n">n_channels</span><span class="p">:</span> <span class="nb">int</span><span class="p">):</span></pre></div>
|
||||
@ -155,7 +153,8 @@ There are pass-through connection at each resolution.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-6'>#</a>
|
||||
</div>
|
||||
<p>First linear layer</p>
|
||||
<p>First linear layer </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">56</span> <span class="bp">self</span><span class="o">.</span><span class="n">lin1</span> <span class="o">=</span> <span class="n">nn</span><span class="o">.</span><span class="n">Linear</span><span class="p">(</span><span class="bp">self</span><span class="o">.</span><span class="n">n_channels</span> <span class="o">//</span> <span class="mi">4</span><span class="p">,</span> <span class="bp">self</span><span class="o">.</span><span class="n">n_channels</span><span class="p">)</span></pre></div>
|
||||
@ -166,7 +165,8 @@ There are pass-through connection at each resolution.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-7'>#</a>
|
||||
</div>
|
||||
<p>Activation</p>
|
||||
<p>Activation </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">58</span> <span class="bp">self</span><span class="o">.</span><span class="n">act</span> <span class="o">=</span> <span class="n">Swish</span><span class="p">()</span></pre></div>
|
||||
@ -177,7 +177,8 @@ There are pass-through connection at each resolution.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-8'>#</a>
|
||||
</div>
|
||||
<p>Second linear layer</p>
|
||||
<p>Second linear layer </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">60</span> <span class="bp">self</span><span class="o">.</span><span class="n">lin2</span> <span class="o">=</span> <span class="n">nn</span><span class="o">.</span><span class="n">Linear</span><span class="p">(</span><span class="bp">self</span><span class="o">.</span><span class="n">n_channels</span><span class="p">,</span> <span class="bp">self</span><span class="o">.</span><span class="n">n_channels</span><span class="p">)</span></pre></div>
|
||||
@ -199,13 +200,9 @@ There are pass-through connection at each resolution.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-10'>#</a>
|
||||
</div>
|
||||
<p>Create sinusoidal position embeddings
|
||||
<a href="../../transformers/positional_encoding.html">same as those from the transformer</a>
|
||||
<script type="math/tex; mode=display">\begin{align}
|
||||
PE^{(1)}_{t,i} &= sin\Bigg(\frac{t}{10000^{\frac{i}{d - 1}}}\Bigg) \\
|
||||
PE^{(2)}_{t,i} &= cos\Bigg(\frac{t}{10000^{\frac{i}{d - 1}}}\Bigg)
|
||||
\end{align}</script>
|
||||
where $d$ is <code>half_dim</code></p>
|
||||
<p>Create sinusoidal position embeddings <a href="../../transformers/positional_encoding.html">same as those from the transformer</a> begin{align} PE^{(1)}_{t,i} &= sinBigg(frac{t}{10000^{frac{i}{d - 1}}}Bigg) \ PE^{(2)}_{t,i} &= cosBigg(frac{t}{10000^{frac{i}{d - 1}}}Bigg) end{align} where <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.69444em;vertical-align:0em;"></span><span class="mord mathnormal">d</span></span></span></span> is <code>half_dim</code>
|
||||
</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">70</span> <span class="n">half_dim</span> <span class="o">=</span> <span class="bp">self</span><span class="o">.</span><span class="n">n_channels</span> <span class="o">//</span> <span class="mi">8</span>
|
||||
@ -220,7 +217,8 @@ where $d$ is <code>half_dim</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-11'>#</a>
|
||||
</div>
|
||||
<p>Transform with the MLP</p>
|
||||
<p>Transform with the MLP </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">77</span> <span class="n">emb</span> <span class="o">=</span> <span class="bp">self</span><span class="o">.</span><span class="n">act</span><span class="p">(</span><span class="bp">self</span><span class="o">.</span><span class="n">lin1</span><span class="p">(</span><span class="n">emb</span><span class="p">))</span>
|
||||
@ -232,7 +230,8 @@ where $d$ is <code>half_dim</code></p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-12'>#</a>
|
||||
</div>
|
||||
|
||||
<p> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">81</span> <span class="k">return</span> <span class="n">emb</span></pre></div>
|
||||
@ -244,8 +243,8 @@ where $d$ is <code>half_dim</code></p>
|
||||
<a href='#section-13'>#</a>
|
||||
</div>
|
||||
<h3>Residual block</h3>
|
||||
<p>A residual block has two convolution layers with group normalization.
|
||||
Each resolution is processed with two residual blocks.</p>
|
||||
<p>A residual block has two convolution layers with group normalization. Each resolution is processed with two residual blocks.</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">84</span><span class="k">class</span> <span class="nc">ResidualBlock</span><span class="p">(</span><span class="n">Module</span><span class="p">):</span></pre></div>
|
||||
@ -256,12 +255,15 @@ Each resolution is processed with two residual blocks.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-14'>#</a>
|
||||
</div>
|
||||
<ul>
|
||||
<li><code>in_channels</code> is the number of input channels</li>
|
||||
<li><code>out_channels</code> is the number of input channels</li>
|
||||
<li><code>time_channels</code> is the number channels in the time step ($t$) embeddings</li>
|
||||
<li><code>n_groups</code> is the number of groups for <a href="../../normalization/group_norm/index.html">group normalization</a></li>
|
||||
</ul>
|
||||
<ul><li><code>in_channels</code>
|
||||
is the number of input channels </li>
|
||||
<li><code>out_channels</code>
|
||||
is the number of input channels </li>
|
||||
<li><code>time_channels</code>
|
||||
is the number channels in the time step (<span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.61508em;vertical-align:0em;"></span><span class="mord mathnormal">t</span></span></span></span>) embeddings </li>
|
||||
<li><code>n_groups</code>
|
||||
is the number of groups for <a href="../../normalization/group_norm/index.html">group normalization</a></li></ul>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">92</span> <span class="k">def</span> <span class="fm">__init__</span><span class="p">(</span><span class="bp">self</span><span class="p">,</span> <span class="n">in_channels</span><span class="p">:</span> <span class="nb">int</span><span class="p">,</span> <span class="n">out_channels</span><span class="p">:</span> <span class="nb">int</span><span class="p">,</span> <span class="n">time_channels</span><span class="p">:</span> <span class="nb">int</span><span class="p">,</span> <span class="n">n_groups</span><span class="p">:</span> <span class="nb">int</span> <span class="o">=</span> <span class="mi">32</span><span class="p">):</span></pre></div>
|
||||
@ -283,7 +285,8 @@ Each resolution is processed with two residual blocks.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-16'>#</a>
|
||||
</div>
|
||||
<p>Group normalization and the first convolution layer</p>
|
||||
<p>Group normalization and the first convolution layer </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">101</span> <span class="bp">self</span><span class="o">.</span><span class="n">norm1</span> <span class="o">=</span> <span class="n">nn</span><span class="o">.</span><span class="n">GroupNorm</span><span class="p">(</span><span class="n">n_groups</span><span class="p">,</span> <span class="n">in_channels</span><span class="p">)</span>
|
||||
@ -296,7 +299,8 @@ Each resolution is processed with two residual blocks.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-17'>#</a>
|
||||
</div>
|
||||
<p>Group normalization and the second convolution layer</p>
|
||||
<p>Group normalization and the second convolution layer </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">106</span> <span class="bp">self</span><span class="o">.</span><span class="n">norm2</span> <span class="o">=</span> <span class="n">nn</span><span class="o">.</span><span class="n">GroupNorm</span><span class="p">(</span><span class="n">n_groups</span><span class="p">,</span> <span class="n">out_channels</span><span class="p">)</span>
|
||||
@ -309,8 +313,8 @@ Each resolution is processed with two residual blocks.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-18'>#</a>
|
||||
</div>
|
||||
<p>If the number of input channels is not equal to the number of output channels we have to
|
||||
project the shortcut connection</p>
|
||||
<p>If the number of input channels is not equal to the number of output channels we have to project the shortcut connection </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">112</span> <span class="k">if</span> <span class="n">in_channels</span> <span class="o">!=</span> <span class="n">out_channels</span><span class="p">:</span>
|
||||
@ -324,7 +328,8 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-19'>#</a>
|
||||
</div>
|
||||
<p>Linear layer for time embeddings</p>
|
||||
<p>Linear layer for time embeddings </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">118</span> <span class="bp">self</span><span class="o">.</span><span class="n">time_emb</span> <span class="o">=</span> <span class="n">nn</span><span class="o">.</span><span class="n">Linear</span><span class="p">(</span><span class="n">time_channels</span><span class="p">,</span> <span class="n">out_channels</span><span class="p">)</span></pre></div>
|
||||
@ -335,10 +340,13 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-20'>#</a>
|
||||
</div>
|
||||
<ul>
|
||||
<li><code>x</code> has shape <code>[batch_size, in_channels, height, width]</code></li>
|
||||
<li><code>t</code> has shape <code>[batch_size, time_channels]</code></li>
|
||||
</ul>
|
||||
<ul><li><code>x</code>
|
||||
has shape <code>[batch_size, in_channels, height, width]</code>
|
||||
</li>
|
||||
<li><code>t</code>
|
||||
has shape <code>[batch_size, time_channels]</code>
|
||||
</li></ul>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">120</span> <span class="k">def</span> <span class="nf">forward</span><span class="p">(</span><span class="bp">self</span><span class="p">,</span> <span class="n">x</span><span class="p">:</span> <span class="n">torch</span><span class="o">.</span><span class="n">Tensor</span><span class="p">,</span> <span class="n">t</span><span class="p">:</span> <span class="n">torch</span><span class="o">.</span><span class="n">Tensor</span><span class="p">):</span></pre></div>
|
||||
@ -349,7 +357,8 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-21'>#</a>
|
||||
</div>
|
||||
<p>First convolution layer</p>
|
||||
<p>First convolution layer </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">126</span> <span class="n">h</span> <span class="o">=</span> <span class="bp">self</span><span class="o">.</span><span class="n">conv1</span><span class="p">(</span><span class="bp">self</span><span class="o">.</span><span class="n">act1</span><span class="p">(</span><span class="bp">self</span><span class="o">.</span><span class="n">norm1</span><span class="p">(</span><span class="n">x</span><span class="p">)))</span></pre></div>
|
||||
@ -360,7 +369,8 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-22'>#</a>
|
||||
</div>
|
||||
<p>Add time embeddings</p>
|
||||
<p>Add time embeddings </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">128</span> <span class="n">h</span> <span class="o">+=</span> <span class="bp">self</span><span class="o">.</span><span class="n">time_emb</span><span class="p">(</span><span class="n">t</span><span class="p">)[:,</span> <span class="p">:,</span> <span class="kc">None</span><span class="p">,</span> <span class="kc">None</span><span class="p">]</span></pre></div>
|
||||
@ -371,7 +381,8 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-23'>#</a>
|
||||
</div>
|
||||
<p>Second convolution layer</p>
|
||||
<p>Second convolution layer </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">130</span> <span class="n">h</span> <span class="o">=</span> <span class="bp">self</span><span class="o">.</span><span class="n">conv2</span><span class="p">(</span><span class="bp">self</span><span class="o">.</span><span class="n">act2</span><span class="p">(</span><span class="bp">self</span><span class="o">.</span><span class="n">norm2</span><span class="p">(</span><span class="n">h</span><span class="p">)))</span></pre></div>
|
||||
@ -382,7 +393,8 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-24'>#</a>
|
||||
</div>
|
||||
<p>Add the shortcut connection and return</p>
|
||||
<p>Add the shortcut connection and return </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">133</span> <span class="k">return</span> <span class="n">h</span> <span class="o">+</span> <span class="bp">self</span><span class="o">.</span><span class="n">shortcut</span><span class="p">(</span><span class="n">x</span><span class="p">)</span></pre></div>
|
||||
@ -395,6 +407,7 @@ project the shortcut connection</p>
|
||||
</div>
|
||||
<h3>Attention block</h3>
|
||||
<p>This is similar to <a href="../../transformers/mha.html">transformer multi-head attention</a>.</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">136</span><span class="k">class</span> <span class="nc">AttentionBlock</span><span class="p">(</span><span class="n">Module</span><span class="p">):</span></pre></div>
|
||||
@ -405,12 +418,15 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-26'>#</a>
|
||||
</div>
|
||||
<ul>
|
||||
<li><code>n_channels</code> is the number of channels in the input</li>
|
||||
<li><code>n_heads</code> is the number of heads in multi-head attention</li>
|
||||
<li><code>d_k</code> is the number of dimensions in each head</li>
|
||||
<li><code>n_groups</code> is the number of groups for <a href="../../normalization/group_norm/index.html">group normalization</a></li>
|
||||
</ul>
|
||||
<ul><li><code>n_channels</code>
|
||||
is the number of channels in the input </li>
|
||||
<li><code>n_heads</code>
|
||||
is the number of heads in multi-head attention </li>
|
||||
<li><code>d_k</code>
|
||||
is the number of dimensions in each head </li>
|
||||
<li><code>n_groups</code>
|
||||
is the number of groups for <a href="../../normalization/group_norm/index.html">group normalization</a></li></ul>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">143</span> <span class="k">def</span> <span class="fm">__init__</span><span class="p">(</span><span class="bp">self</span><span class="p">,</span> <span class="n">n_channels</span><span class="p">:</span> <span class="nb">int</span><span class="p">,</span> <span class="n">n_heads</span><span class="p">:</span> <span class="nb">int</span> <span class="o">=</span> <span class="mi">1</span><span class="p">,</span> <span class="n">d_k</span><span class="p">:</span> <span class="nb">int</span> <span class="o">=</span> <span class="kc">None</span><span class="p">,</span> <span class="n">n_groups</span><span class="p">:</span> <span class="nb">int</span> <span class="o">=</span> <span class="mi">32</span><span class="p">):</span></pre></div>
|
||||
@ -432,7 +448,9 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-28'>#</a>
|
||||
</div>
|
||||
<p>Default <code>d_k</code></p>
|
||||
<p>Default <code>d_k</code>
|
||||
</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">153</span> <span class="k">if</span> <span class="n">d_k</span> <span class="ow">is</span> <span class="kc">None</span><span class="p">:</span>
|
||||
@ -444,7 +462,8 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-29'>#</a>
|
||||
</div>
|
||||
<p>Normalization layer</p>
|
||||
<p>Normalization layer </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">156</span> <span class="bp">self</span><span class="o">.</span><span class="n">norm</span> <span class="o">=</span> <span class="n">nn</span><span class="o">.</span><span class="n">GroupNorm</span><span class="p">(</span><span class="n">n_groups</span><span class="p">,</span> <span class="n">n_channels</span><span class="p">)</span></pre></div>
|
||||
@ -455,7 +474,8 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-30'>#</a>
|
||||
</div>
|
||||
<p>Projections for query, key and values</p>
|
||||
<p>Projections for query, key and values </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">158</span> <span class="bp">self</span><span class="o">.</span><span class="n">projection</span> <span class="o">=</span> <span class="n">nn</span><span class="o">.</span><span class="n">Linear</span><span class="p">(</span><span class="n">n_channels</span><span class="p">,</span> <span class="n">n_heads</span> <span class="o">*</span> <span class="n">d_k</span> <span class="o">*</span> <span class="mi">3</span><span class="p">)</span></pre></div>
|
||||
@ -466,7 +486,8 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-31'>#</a>
|
||||
</div>
|
||||
<p>Linear layer for final transformation</p>
|
||||
<p>Linear layer for final transformation </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">160</span> <span class="bp">self</span><span class="o">.</span><span class="n">output</span> <span class="o">=</span> <span class="n">nn</span><span class="o">.</span><span class="n">Linear</span><span class="p">(</span><span class="n">n_heads</span> <span class="o">*</span> <span class="n">d_k</span><span class="p">,</span> <span class="n">n_channels</span><span class="p">)</span></pre></div>
|
||||
@ -477,7 +498,8 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-32'>#</a>
|
||||
</div>
|
||||
<p>Scale for dot-product attention</p>
|
||||
<p>Scale for dot-product attention </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">162</span> <span class="bp">self</span><span class="o">.</span><span class="n">scale</span> <span class="o">=</span> <span class="n">d_k</span> <span class="o">**</span> <span class="o">-</span><span class="mf">0.5</span></pre></div>
|
||||
@ -488,7 +510,8 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-33'>#</a>
|
||||
</div>
|
||||
|
||||
<p> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">164</span> <span class="bp">self</span><span class="o">.</span><span class="n">n_heads</span> <span class="o">=</span> <span class="n">n_heads</span>
|
||||
@ -500,10 +523,13 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-34'>#</a>
|
||||
</div>
|
||||
<ul>
|
||||
<li><code>x</code> has shape <code>[batch_size, in_channels, height, width]</code></li>
|
||||
<li><code>t</code> has shape <code>[batch_size, time_channels]</code></li>
|
||||
</ul>
|
||||
<ul><li><code>x</code>
|
||||
has shape <code>[batch_size, in_channels, height, width]</code>
|
||||
</li>
|
||||
<li><code>t</code>
|
||||
has shape <code>[batch_size, time_channels]</code>
|
||||
</li></ul>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">167</span> <span class="k">def</span> <span class="nf">forward</span><span class="p">(</span><span class="bp">self</span><span class="p">,</span> <span class="n">x</span><span class="p">:</span> <span class="n">torch</span><span class="o">.</span><span class="n">Tensor</span><span class="p">,</span> <span class="n">t</span><span class="p">:</span> <span class="n">Optional</span><span class="p">[</span><span class="n">torch</span><span class="o">.</span><span class="n">Tensor</span><span class="p">]</span> <span class="o">=</span> <span class="kc">None</span><span class="p">):</span></pre></div>
|
||||
@ -514,8 +540,10 @@ project the shortcut connection</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-35'>#</a>
|
||||
</div>
|
||||
<p><code>t</code> is not used, but it’s kept in the arguments because for the attention layer function signature
|
||||
to match with <code>ResidualBlock</code>.</p>
|
||||
<p><code>t</code>
|
||||
is not used, but it's kept in the arguments because for the attention layer function signature to match with <code>ResidualBlock</code>
|
||||
. </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">174</span> <span class="n">_</span> <span class="o">=</span> <span class="n">t</span></pre></div>
|
||||
@ -526,7 +554,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-36'>#</a>
|
||||
</div>
|
||||
<p>Get shape</p>
|
||||
<p>Get shape </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">176</span> <span class="n">batch_size</span><span class="p">,</span> <span class="n">n_channels</span><span class="p">,</span> <span class="n">height</span><span class="p">,</span> <span class="n">width</span> <span class="o">=</span> <span class="n">x</span><span class="o">.</span><span class="n">shape</span></pre></div>
|
||||
@ -537,7 +566,10 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-37'>#</a>
|
||||
</div>
|
||||
<p>Change <code>x</code> to shape <code>[batch_size, seq, n_channels]</code></p>
|
||||
<p>Change <code>x</code>
|
||||
to shape <code>[batch_size, seq, n_channels]</code>
|
||||
</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">178</span> <span class="n">x</span> <span class="o">=</span> <span class="n">x</span><span class="o">.</span><span class="n">view</span><span class="p">(</span><span class="n">batch_size</span><span class="p">,</span> <span class="n">n_channels</span><span class="p">,</span> <span class="o">-</span><span class="mi">1</span><span class="p">)</span><span class="o">.</span><span class="n">permute</span><span class="p">(</span><span class="mi">0</span><span class="p">,</span> <span class="mi">2</span><span class="p">,</span> <span class="mi">1</span><span class="p">)</span></pre></div>
|
||||
@ -548,7 +580,9 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-38'>#</a>
|
||||
</div>
|
||||
<p>Get query, key, and values (concatenated) and shape it to <code>[batch_size, seq, n_heads, 3 * d_k]</code></p>
|
||||
<p>Get query, key, and values (concatenated) and shape it to <code>[batch_size, seq, n_heads, 3 * d_k]</code>
|
||||
</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">180</span> <span class="n">qkv</span> <span class="o">=</span> <span class="bp">self</span><span class="o">.</span><span class="n">projection</span><span class="p">(</span><span class="n">x</span><span class="p">)</span><span class="o">.</span><span class="n">view</span><span class="p">(</span><span class="n">batch_size</span><span class="p">,</span> <span class="o">-</span><span class="mi">1</span><span class="p">,</span> <span class="bp">self</span><span class="o">.</span><span class="n">n_heads</span><span class="p">,</span> <span class="mi">3</span> <span class="o">*</span> <span class="bp">self</span><span class="o">.</span><span class="n">d_k</span><span class="p">)</span></pre></div>
|
||||
@ -559,7 +593,9 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-39'>#</a>
|
||||
</div>
|
||||
<p>Split query, key, and values. Each of them will have shape <code>[batch_size, seq, n_heads, d_k]</code></p>
|
||||
<p>Split query, key, and values. Each of them will have shape <code>[batch_size, seq, n_heads, d_k]</code>
|
||||
</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">182</span> <span class="n">q</span><span class="p">,</span> <span class="n">k</span><span class="p">,</span> <span class="n">v</span> <span class="o">=</span> <span class="n">torch</span><span class="o">.</span><span class="n">chunk</span><span class="p">(</span><span class="n">qkv</span><span class="p">,</span> <span class="mi">3</span><span class="p">,</span> <span class="n">dim</span><span class="o">=-</span><span class="mi">1</span><span class="p">)</span></pre></div>
|
||||
@ -570,7 +606,19 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-40'>#</a>
|
||||
</div>
|
||||
<p>Calculate scaled dot-product $\frac{Q K^\top}{\sqrt{d_k}}$</p>
|
||||
<p>Calculate scaled dot-product <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:1.633028em;vertical-align:-0.538em;"></span><span class="mord"><span class="mopen nulldelimiter"></span><span class="mfrac"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:1.095028em;"><span style="top:-2.5864385em;"><span class="pstrut" style="height:3em;"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight"><span class="mord sqrt mtight"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.8622307142857143em;"><span class="svg-align" style="top:-3em;"><span class="pstrut" style="height:3em;"></span><span class="mord mtight" style="padding-left:0.833em;"><span class="mord mtight"><span class="mord mathnormal mtight">d</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.3448em;"><span style="top:-2.3487714285714287em;margin-left:0em;margin-right:0.07142857142857144em;"><span class="pstrut" style="height:2.5em;"></span><span class="sizing reset-size3 size1 mtight"><span class="mord mathnormal mtight" style="margin-right:0.03148em;">k</span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.15122857142857138em;"><span></span></span></span></span></span></span></span></span><span style="top:-2.8222307142857144em;"><span class="pstrut" style="height:3em;"></span><span class="hide-tail mtight" style="min-width:0.853em;height:1.08em;"><svg xmlns="http://www.w3.org/2000/svg" width='400em' height='1.08em' viewBox='0 0 400000 1080' preserveAspectRatio='xMinYMin slice'><path d='M95,702
|
||||
c-2.7,0,-7.17,-2.7,-13.5,-8c-5.8,-5.3,-9.5,-10,-9.5,-14
|
||||
c0,-2,0.3,-3.3,1,-4c1.3,-2.7,23.83,-20.7,67.5,-54
|
||||
c44.2,-33.3,65.8,-50.3,66.5,-51c1.3,-1.3,3,-2,5,-2c4.7,0,8.7,3.3,12,10
|
||||
s173,378,173,378c0.7,0,35.3,-71,104,-213c68.7,-142,137.5,-285,206.5,-429
|
||||
c69,-144,104.5,-217.7,106.5,-221
|
||||
l0 -0
|
||||
c5.3,-9.3,12,-14,20,-14
|
||||
H400000v40H845.2724
|
||||
s-225.272,467,-225.272,467s-235,486,-235,486c-2.7,4.7,-9,7,-19,7
|
||||
c-6,0,-10,-1,-12,-3s-194,-422,-194,-422s-65,47,-65,47z
|
||||
M834 80h400000v40h-400000z'/></svg></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.17776928571428574em;"><span></span></span></span></span></span></span></span></span><span style="top:-3.23em;"><span class="pstrut" style="height:3em;"></span><span class="frac-line" style="border-bottom-width:0.04em;"></span></span><span style="top:-3.446108em;"><span class="pstrut" style="height:3em;"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight"><span class="mord mathnormal mtight">Q</span><span class="mord mtight"><span class="mord mathnormal mtight" style="margin-right:0.07153em;">K</span><span class="msupsub"><span class="vlist-t"><span class="vlist-r"><span class="vlist" style="height:0.9270285714285713em;"><span style="top:-2.931em;margin-right:0.07142857142857144em;"><span class="pstrut" style="height:2.5em;"></span><span class="sizing reset-size3 size1 mtight"><span class="mord mtight">⊤</span></span></span></span></span></span></span></span></span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.538em;"><span></span></span></span></span></span><span class="mclose nulldelimiter"></span></span></span></span></span> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">184</span> <span class="n">attn</span> <span class="o">=</span> <span class="n">torch</span><span class="o">.</span><span class="n">einsum</span><span class="p">(</span><span class="s1">'bihd,bjhd->bijh'</span><span class="p">,</span> <span class="n">q</span><span class="p">,</span> <span class="n">k</span><span class="p">)</span> <span class="o">*</span> <span class="bp">self</span><span class="o">.</span><span class="n">scale</span></pre></div>
|
||||
@ -581,7 +629,19 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-41'>#</a>
|
||||
</div>
|
||||
<p>Softmax along the sequence dimension $\underset{seq}{softmax}\Bigg(\frac{Q K^\top}{\sqrt{d_k}}\Bigg)$</p>
|
||||
<p>Softmax along the sequence dimension <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:3.0000299999999998em;vertical-align:-1.25003em;"></span><span class="mord"><span class="mop op-limits"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.6944399999999998em;"><span style="top:-2.20556em;margin-left:0em;"><span class="pstrut" style="height:3em;"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight"><span class="mord mathnormal mtight">se</span><span class="mord mathnormal mtight" style="margin-right:0.03588em;">q</span></span></span></span><span style="top:-3em;"><span class="pstrut" style="height:3em;"></span><span><span class="mop"><span class="mord mathnormal">so</span><span class="mord mathnormal" style="margin-right:0.10764em;">f</span><span class="mord mathnormal">t</span><span class="mord mathnormal">ma</span><span class="mord mathnormal">x</span></span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:1.030548em;"><span></span></span></span></span></span></span><span class="mord"><span class="delimsizing size4">(</span></span><span class="mord"><span class="mopen nulldelimiter"></span><span class="mfrac"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:1.095028em;"><span style="top:-2.5864385em;"><span class="pstrut" style="height:3em;"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight"><span class="mord sqrt mtight"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.8622307142857143em;"><span class="svg-align" style="top:-3em;"><span class="pstrut" style="height:3em;"></span><span class="mord mtight" style="padding-left:0.833em;"><span class="mord mtight"><span class="mord mathnormal mtight">d</span><span class="msupsub"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.3448em;"><span style="top:-2.3487714285714287em;margin-left:0em;margin-right:0.07142857142857144em;"><span class="pstrut" style="height:2.5em;"></span><span class="sizing reset-size3 size1 mtight"><span class="mord mathnormal mtight" style="margin-right:0.03148em;">k</span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.15122857142857138em;"><span></span></span></span></span></span></span></span></span><span style="top:-2.8222307142857144em;"><span class="pstrut" style="height:3em;"></span><span class="hide-tail mtight" style="min-width:0.853em;height:1.08em;"><svg xmlns="http://www.w3.org/2000/svg" width='400em' height='1.08em' viewBox='0 0 400000 1080' preserveAspectRatio='xMinYMin slice'><path d='M95,702
|
||||
c-2.7,0,-7.17,-2.7,-13.5,-8c-5.8,-5.3,-9.5,-10,-9.5,-14
|
||||
c0,-2,0.3,-3.3,1,-4c1.3,-2.7,23.83,-20.7,67.5,-54
|
||||
c44.2,-33.3,65.8,-50.3,66.5,-51c1.3,-1.3,3,-2,5,-2c4.7,0,8.7,3.3,12,10
|
||||
s173,378,173,378c0.7,0,35.3,-71,104,-213c68.7,-142,137.5,-285,206.5,-429
|
||||
c69,-144,104.5,-217.7,106.5,-221
|
||||
l0 -0
|
||||
c5.3,-9.3,12,-14,20,-14
|
||||
H400000v40H845.2724
|
||||
s-225.272,467,-225.272,467s-235,486,-235,486c-2.7,4.7,-9,7,-19,7
|
||||
c-6,0,-10,-1,-12,-3s-194,-422,-194,-422s-65,47,-65,47z
|
||||
M834 80h400000v40h-400000z'/></svg></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.17776928571428574em;"><span></span></span></span></span></span></span></span></span><span style="top:-3.23em;"><span class="pstrut" style="height:3em;"></span><span class="frac-line" style="border-bottom-width:0.04em;"></span></span><span style="top:-3.446108em;"><span class="pstrut" style="height:3em;"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight"><span class="mord mathnormal mtight">Q</span><span class="mord mtight"><span class="mord mathnormal mtight" style="margin-right:0.07153em;">K</span><span class="msupsub"><span class="vlist-t"><span class="vlist-r"><span class="vlist" style="height:0.9270285714285713em;"><span style="top:-2.931em;margin-right:0.07142857142857144em;"><span class="pstrut" style="height:2.5em;"></span><span class="sizing reset-size3 size1 mtight"><span class="mord mtight">⊤</span></span></span></span></span></span></span></span></span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.538em;"><span></span></span></span></span></span><span class="mclose nulldelimiter"></span></span><span class="mord"><span class="delimsizing size4">)</span></span></span></span></span> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">186</span> <span class="n">attn</span> <span class="o">=</span> <span class="n">attn</span><span class="o">.</span><span class="n">softmax</span><span class="p">(</span><span class="n">dim</span><span class="o">=</span><span class="mi">1</span><span class="p">)</span></pre></div>
|
||||
@ -592,7 +652,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-42'>#</a>
|
||||
</div>
|
||||
<p>Multiply by values</p>
|
||||
<p>Multiply by values </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">188</span> <span class="n">res</span> <span class="o">=</span> <span class="n">torch</span><span class="o">.</span><span class="n">einsum</span><span class="p">(</span><span class="s1">'bijh,bjhd->bihd'</span><span class="p">,</span> <span class="n">attn</span><span class="p">,</span> <span class="n">v</span><span class="p">)</span></pre></div>
|
||||
@ -603,7 +664,9 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-43'>#</a>
|
||||
</div>
|
||||
<p>Reshape to <code>[batch_size, seq, n_heads * d_k]</code></p>
|
||||
<p>Reshape to <code>[batch_size, seq, n_heads * d_k]</code>
|
||||
</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">190</span> <span class="n">res</span> <span class="o">=</span> <span class="n">res</span><span class="o">.</span><span class="n">view</span><span class="p">(</span><span class="n">batch_size</span><span class="p">,</span> <span class="o">-</span><span class="mi">1</span><span class="p">,</span> <span class="bp">self</span><span class="o">.</span><span class="n">n_heads</span> <span class="o">*</span> <span class="bp">self</span><span class="o">.</span><span class="n">d_k</span><span class="p">)</span></pre></div>
|
||||
@ -614,7 +677,9 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-44'>#</a>
|
||||
</div>
|
||||
<p>Transform to <code>[batch_size, seq, n_channels]</code></p>
|
||||
<p>Transform to <code>[batch_size, seq, n_channels]</code>
|
||||
</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">192</span> <span class="n">res</span> <span class="o">=</span> <span class="bp">self</span><span class="o">.</span><span class="n">output</span><span class="p">(</span><span class="n">res</span><span class="p">)</span></pre></div>
|
||||
@ -625,7 +690,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-45'>#</a>
|
||||
</div>
|
||||
<p>Add skip connection</p>
|
||||
<p>Add skip connection </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">195</span> <span class="n">res</span> <span class="o">+=</span> <span class="n">x</span></pre></div>
|
||||
@ -636,7 +702,9 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-46'>#</a>
|
||||
</div>
|
||||
<p>Change to shape <code>[batch_size, in_channels, height, width]</code></p>
|
||||
<p>Change to shape <code>[batch_size, in_channels, height, width]</code>
|
||||
</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">198</span> <span class="n">res</span> <span class="o">=</span> <span class="n">res</span><span class="o">.</span><span class="n">permute</span><span class="p">(</span><span class="mi">0</span><span class="p">,</span> <span class="mi">2</span><span class="p">,</span> <span class="mi">1</span><span class="p">)</span><span class="o">.</span><span class="n">view</span><span class="p">(</span><span class="n">batch_size</span><span class="p">,</span> <span class="n">n_channels</span><span class="p">,</span> <span class="n">height</span><span class="p">,</span> <span class="n">width</span><span class="p">)</span></pre></div>
|
||||
@ -647,7 +715,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-47'>#</a>
|
||||
</div>
|
||||
|
||||
<p> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">201</span> <span class="k">return</span> <span class="n">res</span></pre></div>
|
||||
@ -659,7 +728,10 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<a href='#section-48'>#</a>
|
||||
</div>
|
||||
<h3>Down block</h3>
|
||||
<p>This combines <code>ResidualBlock</code> and <code>AttentionBlock</code>. These are used in the first half of U-Net at each resolution.</p>
|
||||
<p>This combines <code>ResidualBlock</code>
|
||||
and <code>AttentionBlock</code>
|
||||
. These are used in the first half of U-Net at each resolution.</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">204</span><span class="k">class</span> <span class="nc">DownBlock</span><span class="p">(</span><span class="n">Module</span><span class="p">):</span></pre></div>
|
||||
@ -702,7 +774,10 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<a href='#section-51'>#</a>
|
||||
</div>
|
||||
<h3>Up block</h3>
|
||||
<p>This combines <code>ResidualBlock</code> and <code>AttentionBlock</code>. These are used in the second half of U-Net at each resolution.</p>
|
||||
<p>This combines <code>ResidualBlock</code>
|
||||
and <code>AttentionBlock</code>
|
||||
. These are used in the second half of U-Net at each resolution.</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">225</span><span class="k">class</span> <span class="nc">UpBlock</span><span class="p">(</span><span class="n">Module</span><span class="p">):</span></pre></div>
|
||||
@ -725,8 +800,9 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-53'>#</a>
|
||||
</div>
|
||||
<p>The input has <code>in_channels + out_channels</code> because we concatenate the output of the same resolution
|
||||
from the first half of the U-Net</p>
|
||||
<p>The input has <code>in_channels + out_channels</code>
|
||||
because we concatenate the output of the same resolution from the first half of the U-Net </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">236</span> <span class="bp">self</span><span class="o">.</span><span class="n">res</span> <span class="o">=</span> <span class="n">ResidualBlock</span><span class="p">(</span><span class="n">in_channels</span> <span class="o">+</span> <span class="n">out_channels</span><span class="p">,</span> <span class="n">out_channels</span><span class="p">,</span> <span class="n">time_channels</span><span class="p">)</span>
|
||||
@ -756,8 +832,11 @@ from the first half of the U-Net</p>
|
||||
<a href='#section-55'>#</a>
|
||||
</div>
|
||||
<h3>Middle block</h3>
|
||||
<p>It combines a <code>ResidualBlock</code>, <code>AttentionBlock</code>, followed by another <code>ResidualBlock</code>.
|
||||
This block is applied at the lowest resolution of the U-Net.</p>
|
||||
<p>It combines a <code>ResidualBlock</code>
|
||||
, <code>AttentionBlock</code>
|
||||
, followed by another <code>ResidualBlock</code>
|
||||
. This block is applied at the lowest resolution of the U-Net.</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">248</span><span class="k">class</span> <span class="nc">MiddleBlock</span><span class="p">(</span><span class="n">Module</span><span class="p">):</span></pre></div>
|
||||
@ -798,7 +877,8 @@ This block is applied at the lowest resolution of the U-Net.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-58'>#</a>
|
||||
</div>
|
||||
<h3>Scale up the feature map by $2 \times$</h3>
|
||||
<h3>Scale up the feature map by <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.72777em;vertical-align:-0.08333em;"></span><span class="mord">2</span><span class="mord">×</span></span></span></span></h3>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">269</span><span class="k">class</span> <span class="nc">Upsample</span><span class="p">(</span><span class="n">nn</span><span class="o">.</span><span class="n">Module</span><span class="p">):</span></pre></div>
|
||||
@ -833,8 +913,10 @@ This block is applied at the lowest resolution of the U-Net.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-61'>#</a>
|
||||
</div>
|
||||
<p><code>t</code> is not used, but it’s kept in the arguments because for the attention layer function signature
|
||||
to match with <code>ResidualBlock</code>.</p>
|
||||
<p><code>t</code>
|
||||
is not used, but it's kept in the arguments because for the attention layer function signature to match with <code>ResidualBlock</code>
|
||||
. </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">281</span> <span class="n">_</span> <span class="o">=</span> <span class="n">t</span>
|
||||
@ -846,7 +928,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-62'>#</a>
|
||||
</div>
|
||||
<h3>Scale down the feature map by $\frac{1}{2} \times$</h3>
|
||||
<h3>Scale down the feature map by <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:1.190108em;vertical-align:-0.345em;"></span><span class="mord"><span class="mopen nulldelimiter"></span><span class="mfrac"><span class="vlist-t vlist-t2"><span class="vlist-r"><span class="vlist" style="height:0.845108em;"><span style="top:-2.6550000000000002em;"><span class="pstrut" style="height:3em;"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight"><span class="mord mtight">2</span></span></span></span><span style="top:-3.23em;"><span class="pstrut" style="height:3em;"></span><span class="frac-line" style="border-bottom-width:0.04em;"></span></span><span style="top:-3.394em;"><span class="pstrut" style="height:3em;"></span><span class="sizing reset-size6 size3 mtight"><span class="mord mtight"><span class="mord mtight">1</span></span></span></span></span><span class="vlist-s"></span></span><span class="vlist-r"><span class="vlist" style="height:0.345em;"><span></span></span></span></span></span><span class="mclose nulldelimiter"></span></span><span class="mord">×</span></span></span></span></h3>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">285</span><span class="k">class</span> <span class="nc">Downsample</span><span class="p">(</span><span class="n">nn</span><span class="o">.</span><span class="n">Module</span><span class="p">):</span></pre></div>
|
||||
@ -881,8 +964,10 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-65'>#</a>
|
||||
</div>
|
||||
<p><code>t</code> is not used, but it’s kept in the arguments because for the attention layer function signature
|
||||
to match with <code>ResidualBlock</code>.</p>
|
||||
<p><code>t</code>
|
||||
is not used, but it's kept in the arguments because for the attention layer function signature to match with <code>ResidualBlock</code>
|
||||
. </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">297</span> <span class="n">_</span> <span class="o">=</span> <span class="n">t</span>
|
||||
@ -895,6 +980,7 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<a href='#section-66'>#</a>
|
||||
</div>
|
||||
<h2>U-Net</h2>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">301</span><span class="k">class</span> <span class="nc">UNet</span><span class="p">(</span><span class="n">Module</span><span class="p">):</span></pre></div>
|
||||
@ -905,13 +991,19 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-67'>#</a>
|
||||
</div>
|
||||
<ul>
|
||||
<li><code>image_channels</code> is the number of channels in the image. $3$ for RGB.</li>
|
||||
<li><code>n_channels</code> is number of channels in the initial feature map that we transform the image into</li>
|
||||
<li><code>ch_mults</code> is the list of channel numbers at each resolution. The number of channels is <code>ch_mults[i] * n_channels</code></li>
|
||||
<li><code>is_attn</code> is a list of booleans that indicate whether to use attention at each resolution</li>
|
||||
<li><code>n_blocks</code> is the number of <code>UpDownBlocks</code> at each resolution</li>
|
||||
</ul>
|
||||
<ul><li><code>image_channels</code>
|
||||
is the number of channels in the image. <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.64444em;vertical-align:0em;"></span><span class="mord">3</span></span></span></span> for RGB. </li>
|
||||
<li><code>n_channels</code>
|
||||
is number of channels in the initial feature map that we transform the image into </li>
|
||||
<li><code>ch_mults</code>
|
||||
is the list of channel numbers at each resolution. The number of channels is <code>ch_mults[i] * n_channels</code>
|
||||
</li>
|
||||
<li><code>is_attn</code>
|
||||
is a list of booleans that indicate whether to use attention at each resolution </li>
|
||||
<li><code>n_blocks</code>
|
||||
is the number of <code>UpDownBlocks</code>
|
||||
at each resolution</li></ul>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">306</span> <span class="k">def</span> <span class="fm">__init__</span><span class="p">(</span><span class="bp">self</span><span class="p">,</span> <span class="n">image_channels</span><span class="p">:</span> <span class="nb">int</span> <span class="o">=</span> <span class="mi">3</span><span class="p">,</span> <span class="n">n_channels</span><span class="p">:</span> <span class="nb">int</span> <span class="o">=</span> <span class="mi">64</span><span class="p">,</span>
|
||||
@ -936,7 +1028,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-69'>#</a>
|
||||
</div>
|
||||
<p>Number of resolutions</p>
|
||||
<p>Number of resolutions </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">320</span> <span class="n">n_resolutions</span> <span class="o">=</span> <span class="nb">len</span><span class="p">(</span><span class="n">ch_mults</span><span class="p">)</span></pre></div>
|
||||
@ -947,7 +1040,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-70'>#</a>
|
||||
</div>
|
||||
<p>Project image into feature map</p>
|
||||
<p>Project image into feature map </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">323</span> <span class="bp">self</span><span class="o">.</span><span class="n">image_proj</span> <span class="o">=</span> <span class="n">nn</span><span class="o">.</span><span class="n">Conv2d</span><span class="p">(</span><span class="n">image_channels</span><span class="p">,</span> <span class="n">n_channels</span><span class="p">,</span> <span class="n">kernel_size</span><span class="o">=</span><span class="p">(</span><span class="mi">3</span><span class="p">,</span> <span class="mi">3</span><span class="p">),</span> <span class="n">padding</span><span class="o">=</span><span class="p">(</span><span class="mi">1</span><span class="p">,</span> <span class="mi">1</span><span class="p">))</span></pre></div>
|
||||
@ -958,7 +1052,9 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-71'>#</a>
|
||||
</div>
|
||||
<p>Time embedding layer. Time embedding has <code>n_channels * 4</code> channels</p>
|
||||
<p>Time embedding layer. Time embedding has <code>n_channels * 4</code>
|
||||
channels </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">326</span> <span class="bp">self</span><span class="o">.</span><span class="n">time_emb</span> <span class="o">=</span> <span class="n">TimeEmbedding</span><span class="p">(</span><span class="n">n_channels</span> <span class="o">*</span> <span class="mi">4</span><span class="p">)</span></pre></div>
|
||||
@ -970,6 +1066,7 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<a href='#section-72'>#</a>
|
||||
</div>
|
||||
<h4>First half of U-Net - decreasing resolution</h4>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">329</span> <span class="n">down</span> <span class="o">=</span> <span class="p">[]</span></pre></div>
|
||||
@ -980,7 +1077,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-73'>#</a>
|
||||
</div>
|
||||
<p>Number of channels</p>
|
||||
<p>Number of channels </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">331</span> <span class="n">out_channels</span> <span class="o">=</span> <span class="n">in_channels</span> <span class="o">=</span> <span class="n">n_channels</span></pre></div>
|
||||
@ -991,7 +1089,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-74'>#</a>
|
||||
</div>
|
||||
<p>For each resolution</p>
|
||||
<p>For each resolution </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">333</span> <span class="k">for</span> <span class="n">i</span> <span class="ow">in</span> <span class="nb">range</span><span class="p">(</span><span class="n">n_resolutions</span><span class="p">):</span></pre></div>
|
||||
@ -1002,7 +1101,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-75'>#</a>
|
||||
</div>
|
||||
<p>Number of output channels at this resolution</p>
|
||||
<p>Number of output channels at this resolution </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">335</span> <span class="n">out_channels</span> <span class="o">=</span> <span class="n">in_channels</span> <span class="o">*</span> <span class="n">ch_mults</span><span class="p">[</span><span class="n">i</span><span class="p">]</span></pre></div>
|
||||
@ -1013,7 +1113,9 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-76'>#</a>
|
||||
</div>
|
||||
<p>Add <code>n_blocks</code></p>
|
||||
<p>Add <code>n_blocks</code>
|
||||
</p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">337</span> <span class="k">for</span> <span class="n">_</span> <span class="ow">in</span> <span class="nb">range</span><span class="p">(</span><span class="n">n_blocks</span><span class="p">):</span>
|
||||
@ -1026,7 +1128,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-77'>#</a>
|
||||
</div>
|
||||
<p>Down sample at all resolutions except the last</p>
|
||||
<p>Down sample at all resolutions except the last </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">341</span> <span class="k">if</span> <span class="n">i</span> <span class="o"><</span> <span class="n">n_resolutions</span> <span class="o">-</span> <span class="mi">1</span><span class="p">:</span>
|
||||
@ -1038,7 +1141,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-78'>#</a>
|
||||
</div>
|
||||
<p>Combine the set of modules</p>
|
||||
<p>Combine the set of modules </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">345</span> <span class="bp">self</span><span class="o">.</span><span class="n">down</span> <span class="o">=</span> <span class="n">nn</span><span class="o">.</span><span class="n">ModuleList</span><span class="p">(</span><span class="n">down</span><span class="p">)</span></pre></div>
|
||||
@ -1049,7 +1153,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-79'>#</a>
|
||||
</div>
|
||||
<p>Middle block</p>
|
||||
<p>Middle block </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">348</span> <span class="bp">self</span><span class="o">.</span><span class="n">middle</span> <span class="o">=</span> <span class="n">MiddleBlock</span><span class="p">(</span><span class="n">out_channels</span><span class="p">,</span> <span class="n">n_channels</span> <span class="o">*</span> <span class="mi">4</span><span class="p">,</span> <span class="p">)</span></pre></div>
|
||||
@ -1061,6 +1166,7 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<a href='#section-80'>#</a>
|
||||
</div>
|
||||
<h4>Second half of U-Net - increasing resolution</h4>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">351</span> <span class="n">up</span> <span class="o">=</span> <span class="p">[]</span></pre></div>
|
||||
@ -1071,7 +1177,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-81'>#</a>
|
||||
</div>
|
||||
<p>Number of channels</p>
|
||||
<p>Number of channels </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">353</span> <span class="n">in_channels</span> <span class="o">=</span> <span class="n">out_channels</span></pre></div>
|
||||
@ -1082,7 +1189,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-82'>#</a>
|
||||
</div>
|
||||
<p>For each resolution</p>
|
||||
<p>For each resolution </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">355</span> <span class="k">for</span> <span class="n">i</span> <span class="ow">in</span> <span class="nb">reversed</span><span class="p">(</span><span class="nb">range</span><span class="p">(</span><span class="n">n_resolutions</span><span class="p">)):</span></pre></div>
|
||||
@ -1093,7 +1201,9 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-83'>#</a>
|
||||
</div>
|
||||
<p><code>n_blocks</code> at the same resolution</p>
|
||||
<p><code>n_blocks</code>
|
||||
at the same resolution </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">357</span> <span class="n">out_channels</span> <span class="o">=</span> <span class="n">in_channels</span>
|
||||
@ -1106,7 +1216,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-84'>#</a>
|
||||
</div>
|
||||
<p>Final block to reduce the number of channels</p>
|
||||
<p>Final block to reduce the number of channels </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">361</span> <span class="n">out_channels</span> <span class="o">=</span> <span class="n">in_channels</span> <span class="o">//</span> <span class="n">ch_mults</span><span class="p">[</span><span class="n">i</span><span class="p">]</span>
|
||||
@ -1119,7 +1230,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-85'>#</a>
|
||||
</div>
|
||||
<p>Up sample at all resolutions except last</p>
|
||||
<p>Up sample at all resolutions except last </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">365</span> <span class="k">if</span> <span class="n">i</span> <span class="o">></span> <span class="mi">0</span><span class="p">:</span>
|
||||
@ -1131,7 +1243,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-86'>#</a>
|
||||
</div>
|
||||
<p>Combine the set of modules</p>
|
||||
<p>Combine the set of modules </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">369</span> <span class="bp">self</span><span class="o">.</span><span class="n">up</span> <span class="o">=</span> <span class="n">nn</span><span class="o">.</span><span class="n">ModuleList</span><span class="p">(</span><span class="n">up</span><span class="p">)</span></pre></div>
|
||||
@ -1142,7 +1255,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-87'>#</a>
|
||||
</div>
|
||||
<p>Final normalization and convolution layer</p>
|
||||
<p>Final normalization and convolution layer </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">372</span> <span class="bp">self</span><span class="o">.</span><span class="n">norm</span> <span class="o">=</span> <span class="n">nn</span><span class="o">.</span><span class="n">GroupNorm</span><span class="p">(</span><span class="mi">8</span><span class="p">,</span> <span class="n">n_channels</span><span class="p">)</span>
|
||||
@ -1155,10 +1269,13 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-88'>#</a>
|
||||
</div>
|
||||
<ul>
|
||||
<li><code>x</code> has shape <code>[batch_size, in_channels, height, width]</code></li>
|
||||
<li><code>t</code> has shape <code>[batch_size]</code></li>
|
||||
</ul>
|
||||
<ul><li><code>x</code>
|
||||
has shape <code>[batch_size, in_channels, height, width]</code>
|
||||
</li>
|
||||
<li><code>t</code>
|
||||
has shape <code>[batch_size]</code>
|
||||
</li></ul>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">376</span> <span class="k">def</span> <span class="nf">forward</span><span class="p">(</span><span class="bp">self</span><span class="p">,</span> <span class="n">x</span><span class="p">:</span> <span class="n">torch</span><span class="o">.</span><span class="n">Tensor</span><span class="p">,</span> <span class="n">t</span><span class="p">:</span> <span class="n">torch</span><span class="o">.</span><span class="n">Tensor</span><span class="p">):</span></pre></div>
|
||||
@ -1169,7 +1286,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-89'>#</a>
|
||||
</div>
|
||||
<p>Get time-step embeddings</p>
|
||||
<p>Get time-step embeddings </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">383</span> <span class="n">t</span> <span class="o">=</span> <span class="bp">self</span><span class="o">.</span><span class="n">time_emb</span><span class="p">(</span><span class="n">t</span><span class="p">)</span></pre></div>
|
||||
@ -1180,7 +1298,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-90'>#</a>
|
||||
</div>
|
||||
<p>Get image projection</p>
|
||||
<p>Get image projection </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">386</span> <span class="n">x</span> <span class="o">=</span> <span class="bp">self</span><span class="o">.</span><span class="n">image_proj</span><span class="p">(</span><span class="n">x</span><span class="p">)</span></pre></div>
|
||||
@ -1191,7 +1310,9 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-91'>#</a>
|
||||
</div>
|
||||
<p><code>h</code> will store outputs at each resolution for skip connection</p>
|
||||
<p><code>h</code>
|
||||
will store outputs at each resolution for skip connection </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">389</span> <span class="n">h</span> <span class="o">=</span> <span class="p">[</span><span class="n">x</span><span class="p">]</span></pre></div>
|
||||
@ -1202,7 +1323,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-92'>#</a>
|
||||
</div>
|
||||
<p>First half of U-Net</p>
|
||||
<p>First half of U-Net </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">391</span> <span class="k">for</span> <span class="n">m</span> <span class="ow">in</span> <span class="bp">self</span><span class="o">.</span><span class="n">down</span><span class="p">:</span>
|
||||
@ -1215,7 +1337,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-93'>#</a>
|
||||
</div>
|
||||
<p>Middle (bottom)</p>
|
||||
<p>Middle (bottom) </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">396</span> <span class="n">x</span> <span class="o">=</span> <span class="bp">self</span><span class="o">.</span><span class="n">middle</span><span class="p">(</span><span class="n">x</span><span class="p">,</span> <span class="n">t</span><span class="p">)</span></pre></div>
|
||||
@ -1226,7 +1349,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-94'>#</a>
|
||||
</div>
|
||||
<p>Second half of U-Net</p>
|
||||
<p>Second half of U-Net </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">399</span> <span class="k">for</span> <span class="n">m</span> <span class="ow">in</span> <span class="bp">self</span><span class="o">.</span><span class="n">up</span><span class="p">:</span>
|
||||
@ -1240,7 +1364,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-95'>#</a>
|
||||
</div>
|
||||
<p>Get the skip connection from first half of U-Net and concatenate</p>
|
||||
<p>Get the skip connection from first half of U-Net and concatenate </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">404</span> <span class="n">s</span> <span class="o">=</span> <span class="n">h</span><span class="o">.</span><span class="n">pop</span><span class="p">()</span>
|
||||
@ -1252,7 +1377,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-96'>#</a>
|
||||
</div>
|
||||
|
||||
<p> </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">407</span> <span class="n">x</span> <span class="o">=</span> <span class="n">m</span><span class="p">(</span><span class="n">x</span><span class="p">,</span> <span class="n">t</span><span class="p">)</span></pre></div>
|
||||
@ -1263,7 +1389,8 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<div class='section-link'>
|
||||
<a href='#section-97'>#</a>
|
||||
</div>
|
||||
<p>Final normalization and convolution</p>
|
||||
<p>Final normalization and convolution </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">410</span> <span class="k">return</span> <span class="bp">self</span><span class="o">.</span><span class="n">final</span><span class="p">(</span><span class="bp">self</span><span class="o">.</span><span class="n">act</span><span class="p">(</span><span class="bp">self</span><span class="o">.</span><span class="n">norm</span><span class="p">(</span><span class="n">x</span><span class="p">)))</span></pre></div>
|
||||
@ -1274,24 +1401,6 @@ to match with <code>ResidualBlock</code>.</p>
|
||||
<a href="https://labml.ai">labml.ai</a>
|
||||
</div>
|
||||
</div>
|
||||
<script src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.4/MathJax.js?config=TeX-AMS_HTML">
|
||||
</script>
|
||||
<!-- MathJax configuration -->
|
||||
<script type="text/x-mathjax-config">
|
||||
MathJax.Hub.Config({
|
||||
tex2jax: {
|
||||
inlineMath: [ ['$','$'] ],
|
||||
displayMath: [ ['$$','$$'] ],
|
||||
processEscapes: true,
|
||||
processEnvironments: true
|
||||
},
|
||||
// Center justify equations in code and markdown cells. Elsewhere
|
||||
// we use CSS to left justify single line equations in code cells.
|
||||
displayAlign: 'center',
|
||||
"HTML-CSS": { fonts: ["TeX"] }
|
||||
});
|
||||
|
||||
</script>
|
||||
<script>
|
||||
function handleImages() {
|
||||
var images = document.querySelectorAll('p>img')
|
||||
|
@ -24,6 +24,8 @@
|
||||
<link rel="shortcut icon" href="/icon.png"/>
|
||||
<link rel="stylesheet" href="../../pylit.css">
|
||||
<link rel="canonical" href="https://nn.labml.ai/diffusion/ddpm/utils.html"/>
|
||||
<link rel="stylesheet" href="https://cdn.jsdelivr.net/npm/katex@0.13.18/dist/katex.min.css" integrity="sha384-zTROYFVGOfTw7JV7KUu8udsvW2fx4lWOsCEDqhBreBwlHI4ioVRtmIvEThzJHGET" crossorigin="anonymous">
|
||||
|
||||
<!-- Global site tag (gtag.js) - Google Analytics -->
|
||||
<script async src="https://www.googletagmanager.com/gtag/js?id=G-4V3HC8HBLH"></script>
|
||||
<script>
|
||||
@ -68,6 +70,7 @@
|
||||
<a href='#section-0'>#</a>
|
||||
</div>
|
||||
<h1>Utility functions for <a href="index.html">DDPM</a> experiemnt</h1>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">10</span><span></span><span class="kn">import</span> <span class="nn">torch.utils.data</span></pre></div>
|
||||
@ -78,7 +81,8 @@
|
||||
<div class='section-link'>
|
||||
<a href='#section-1'>#</a>
|
||||
</div>
|
||||
<p>Gather consts for $t$ and reshape to feature map shape</p>
|
||||
<p>Gather consts for <span class="katex"><span class="katex-html" aria-hidden="true"><span class="base"><span class="strut" style="height:0.61508em;vertical-align:0em;"></span><span class="mord mathnormal">t</span></span></span></span> and reshape to feature map shape </p>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre><span class="lineno">13</span><span class="k">def</span> <span class="nf">gather</span><span class="p">(</span><span class="n">consts</span><span class="p">:</span> <span class="n">torch</span><span class="o">.</span><span class="n">Tensor</span><span class="p">,</span> <span class="n">t</span><span class="p">:</span> <span class="n">torch</span><span class="o">.</span><span class="n">Tensor</span><span class="p">):</span></pre></div>
|
||||
@ -101,24 +105,6 @@
|
||||
<a href="https://labml.ai">labml.ai</a>
|
||||
</div>
|
||||
</div>
|
||||
<script src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.4/MathJax.js?config=TeX-AMS_HTML">
|
||||
</script>
|
||||
<!-- MathJax configuration -->
|
||||
<script type="text/x-mathjax-config">
|
||||
MathJax.Hub.Config({
|
||||
tex2jax: {
|
||||
inlineMath: [ ['$','$'] ],
|
||||
displayMath: [ ['$$','$$'] ],
|
||||
processEscapes: true,
|
||||
processEnvironments: true
|
||||
},
|
||||
// Center justify equations in code and markdown cells. Elsewhere
|
||||
// we use CSS to left justify single line equations in code cells.
|
||||
displayAlign: 'center',
|
||||
"HTML-CSS": { fonts: ["TeX"] }
|
||||
});
|
||||
|
||||
</script>
|
||||
<script>
|
||||
function handleImages() {
|
||||
var images = document.querySelectorAll('p>img')
|
||||
|
@ -24,6 +24,8 @@
|
||||
<link rel="shortcut icon" href="/icon.png"/>
|
||||
<link rel="stylesheet" href="../pylit.css">
|
||||
<link rel="canonical" href="https://nn.labml.ai/diffusion/index.html"/>
|
||||
<link rel="stylesheet" href="https://cdn.jsdelivr.net/npm/katex@0.13.18/dist/katex.min.css" integrity="sha384-zTROYFVGOfTw7JV7KUu8udsvW2fx4lWOsCEDqhBreBwlHI4ioVRtmIvEThzJHGET" crossorigin="anonymous">
|
||||
|
||||
<!-- Global site tag (gtag.js) - Google Analytics -->
|
||||
<script async src="https://www.googletagmanager.com/gtag/js?id=G-4V3HC8HBLH"></script>
|
||||
<script>
|
||||
@ -67,9 +69,8 @@
|
||||
<a href='#section-0'>#</a>
|
||||
</div>
|
||||
<h1>Diffusion models</h1>
|
||||
<ul>
|
||||
<li><a href="ddpm/index.html">Denoising Diffusion Probabilistic Models (DDPM)</a></li>
|
||||
</ul>
|
||||
<ul><li><a href="ddpm/index.html">Denoising Diffusion Probabilistic Models (DDPM)</a></li></ul>
|
||||
|
||||
</div>
|
||||
<div class='code'>
|
||||
<div class="highlight"><pre></pre></div>
|
||||
@ -80,24 +81,6 @@
|
||||
<a href="https://labml.ai">labml.ai</a>
|
||||
</div>
|
||||
</div>
|
||||
<script src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.4/MathJax.js?config=TeX-AMS_HTML">
|
||||
</script>
|
||||
<!-- MathJax configuration -->
|
||||
<script type="text/x-mathjax-config">
|
||||
MathJax.Hub.Config({
|
||||
tex2jax: {
|
||||
inlineMath: [ ['$','$'] ],
|
||||
displayMath: [ ['$$','$$'] ],
|
||||
processEscapes: true,
|
||||
processEnvironments: true
|
||||
},
|
||||
// Center justify equations in code and markdown cells. Elsewhere
|
||||
// we use CSS to left justify single line equations in code cells.
|
||||
displayAlign: 'center',
|
||||
"HTML-CSS": { fonts: ["TeX"] }
|
||||
});
|
||||
|
||||
</script>
|
||||
<script>
|
||||
function handleImages() {
|
||||
var images = document.querySelectorAll('p>img')
|
||||
|
Reference in New Issue
Block a user