\relax 
\providecommand\hyper@newdestlabel[2]{}
\providecommand\HyField@AuxAddToFields[1]{}
\providecommand\HyField@AuxAddToCoFields[2]{}
\@writefile{toc}{\contentsline {section}{\numberline {1}Where We Are: Three Things You Already Have}{1}{section.1}\protected@file@percent }
\newlabel{sec:recap}{{1}{1}{Where We Are: Three Things You Already Have}{section.1}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {1.1}The autoencoder of HW5 Problem 1}{1}{subsection.1.1}\protected@file@percent }
\@writefile{toc}{\contentsline {subsection}{\numberline {1.2}PCA is a linear autoencoder (HW3)}{1}{subsection.1.2}\protected@file@percent }
\@writefile{toc}{\contentsline {subsection}{\numberline {1.3}A generative model with a latent variable (HW4)}{1}{subsection.1.3}\protected@file@percent }
\@writefile{toc}{\contentsline {subsection}{\numberline {1.4}What an autoencoder cannot do}{2}{subsection.1.4}\protected@file@percent }
\@writefile{toc}{\contentsline {section}{\numberline {2}The Big Picture: an Autoencoder Whose Code Is Random}{2}{section.2}\protected@file@percent }
\newlabel{sec:overview}{{2}{2}{The Big Picture: an Autoencoder Whose Code Is Random}{section.2}{}}
\@writefile{toc}{\contentsline {section}{\numberline {3}A Continuous Latent Variable}{2}{section.3}\protected@file@percent }
\newlabel{sec:latent}{{3}{2}{A Continuous Latent Variable}{section.3}{}}
\newlabel{eq:marginal}{{1}{2}{A Continuous Latent Variable}{equation.1}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {1}{\ignorespaces The VAE's layers (HW5 sizes; only a few units of each layer are drawn). Lines between columns are ordinary dense connections with weights. The colored arrows into the code $\mathbf  {z}$ are \emph  {not} weights: each code unit is computed from its own $\mu _j$, $\log \sigma _j^2$ and noise $\epsilon _j$ (one color per latent dimension). About $631{,}000$ weights in total, almost all in the first and last layer.}}{3}{figure.1}\protected@file@percent }
\newlabel{fig:vae-arch}{{1}{3}{The VAE's layers (HW5 sizes; only a few units of each layer are drawn). Lines between columns are ordinary dense connections with weights. The colored arrows into the code $\bz $ are \emph {not} weights: each code unit is computed from its own $\mu _j$, $\log \sigma _j^2$ and noise $\epsilon _j$ (one color per latent dimension). About $631{,}000$ weights in total, almost all in the first and last layer}{figure.1}{}}
\@writefile{toc}{\contentsline {paragraph}{Now make the decoder a neural network.}{4}{section*.1}\protected@file@percent }
\@writefile{toc}{\contentsline {section}{\numberline {4}Comparing Two Distributions: KL Divergence}{4}{section.4}\protected@file@percent }
\newlabel{sec:kl}{{4}{4}{Comparing Two Distributions: KL Divergence}{section.4}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {4.1}Definition, with a discrete example}{4}{subsection.4.1}\protected@file@percent }
\@writefile{toc}{\contentsline {subsection}{\numberline {4.2}You have been minimizing KL all along}{5}{subsection.4.2}\protected@file@percent }
\@writefile{toc}{\contentsline {subsection}{\numberline {4.3}KL between two Gaussians}{5}{subsection.4.3}\protected@file@percent }
\newlabel{eq:klclosed}{{2}{5}{KL between two Gaussians}{equation.2}{}}
\newlabel{eq:klsum}{{3}{5}{KL between two Gaussians}{equation.3}{}}
\@writefile{toc}{\contentsline {section}{\numberline {5}The Evidence Lower Bound (ELBO)}{6}{section.5}\protected@file@percent }
\newlabel{sec:elbo}{{5}{6}{The Evidence Lower Bound (ELBO)}{section.5}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {5.1}One identity, true for any guess $q$}{6}{subsection.5.1}\protected@file@percent }
\newlabel{eq:decomp}{{4}{6}{One identity, true for any guess $q$}{equation.4}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {5.2}This is EM}{6}{subsection.5.2}\protected@file@percent }
\@writefile{toc}{\contentsline {subsection}{\numberline {5.3}The form the VAE uses}{6}{subsection.5.3}\protected@file@percent }
\newlabel{eq:elbo}{{5}{6}{The form the VAE uses}{equation.5}{}}
\@writefile{toc}{\contentsline {section}{\numberline {6}From the ELBO to a Network}{7}{section.6}\protected@file@percent }
\newlabel{sec:network}{{6}{7}{From the ELBO to a Network}{section.6}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {2}{\ignorespaces The VAE at a glance. \textbf  {Training:} the encoder turns an image into a Gaussian guess $\mathcal  {N}(\mu ,\sigma ^2)$ for its code; a code $\mathbf  {z}$ is sampled from that guess (with the randomness moved into the input $\epsilon $); the decoder turns $\mathbf  {z}$ into pixel probabilities $\hat  \mathbf  {x}$. The loss has two terms with two jobs, and one backward pass updates both networks. \textbf  {Generation:} throw the encoder away, draw $\mathbf  {z}$ from the prior, decode.}}{7}{figure.2}\protected@file@percent }
\newlabel{fig:vae-pipeline}{{2}{7}{The VAE at a glance. \textbf {Training:} the encoder turns an image into a Gaussian guess $\N (\mu ,\sigma ^2)$ for its code; a code $\bz $ is sampled from that guess (with the randomness moved into the input $\epsilon $); the decoder turns $\bz $ into pixel probabilities $\hat \bx $. The loss has two terms with two jobs, and one backward pass updates both networks. \textbf {Generation:} throw the encoder away, draw $\bz $ from the prior, decode}{figure.2}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {6.1}Amortized inference: one network that guesses every posterior}{7}{subsection.6.1}\protected@file@percent }
\@writefile{toc}{\contentsline {subsection}{\numberline {6.2}How new is this, structurally?}{7}{subsection.6.2}\protected@file@percent }
\newlabel{sec:structure}{{6.2}{7}{How new is this, structurally?}{subsection.6.2}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {6.3}The reconstruction term is a Week~9 loss}{8}{subsection.6.3}\protected@file@percent }
\@writefile{toc}{\contentsline {subsection}{\numberline {6.4}The loss for one image}{8}{subsection.6.4}\protected@file@percent }
\newlabel{eq:loss}{{6}{8}{The loss for one image}{equation.6}{}}
\@writefile{toc}{\contentsline {section}{\numberline {7}Training Through a Random Node: the Reparameterization Trick}{9}{section.7}\protected@file@percent }
\newlabel{sec:reparam}{{7}{9}{Training Through a Random Node: the Reparameterization Trick}{section.7}{}}
\@writefile{toc}{\contentsline {subsection}{\numberline {7.1}The problem}{9}{subsection.7.1}\protected@file@percent }
\@writefile{toc}{\contentsline {subsection}{\numberline {7.2}The trick: move the randomness into an input}{9}{subsection.7.2}\protected@file@percent }
\newlabel{eq:reparam}{{7}{9}{The trick: move the randomness into an input}{equation.7}{}}
\@writefile{toc}{\contentsline {section}{\numberline {8}What the KL Term Buys: Autoencoder vs.\ VAE (demo)}{10}{section.8}\protected@file@percent }
\newlabel{sec:demo}{{8}{10}{What the KL Term Buys: Autoencoder vs.\ VAE (demo)}{section.8}{}}
\@writefile{lof}{\contentsline {figure}{\numberline {3}{\ignorespaces Codes of all 1797 images, colored by digit. Left: the autoencoder's codes, with arbitrary scale (standard deviation about $8$ and $5$) and uneven spread. Right: the VAE's means, pulled into the region where $\mathcal  {N}(0,I)$ lives.}}{10}{figure.3}\protected@file@percent }
\newlabel{fig:latent}{{3}{10}{Codes of all 1797 images, colored by digit. Left: the autoencoder's codes, with arbitrary scale (standard deviation about $8$ and $5$) and uneven spread. Right: the VAE's means, pulled into the region where $\N (0,I)$ lives}{figure.3}{}}
\@writefile{toc}{\contentsline {section}{\numberline {9}EM vs.\ VAE: the Same Idea, Two Implementations}{11}{section.9}\protected@file@percent }
\newlabel{sec:em-bridge}{{9}{11}{EM vs.\ VAE: the Same Idea, Two Implementations}{section.9}{}}
\@writefile{toc}{\contentsline {section}{\numberline {10}HW5 Problem 3: Map and Implementation Notes}{12}{section.10}\protected@file@percent }
\newlabel{sec:hw5}{{10}{12}{HW5 Problem 3: Map and Implementation Notes}{section.10}{}}
\@writefile{toc}{\contentsline {section}{\numberline {11}Beyond HW5 (optional)}{12}{section.11}\protected@file@percent }
\gdef \@abspage@last{13}
