import nbformat as nbf nb = nbf.v4.new_notebook(); C=[] md = lambda s: C.append(nbf.v4.new_markdown_cell(s)) code = lambda s: C.append(nbf.v4.new_code_cell(s)) md("""# GAN lecture demo (Week 10) — CS6140 Companion to the lecture note **GAN_claude_0929.pdf**. Everything here is **1-D**, so every picture is a histogram you can compare with the board. (HW5 Problem 4 is the 2-D version on `2gaussian.txt`; nothing here is its solution.) | Cell | Board segment | What it shows | |---|---|---| | 1 | setup | imports, device (tiny models: CPU) | | 2 | "a generator is a sampler" | push-forward: $x=G(z)$ turns $\\mathcal N(0,1)$ noise into other distributions | | 3 | optimal discriminator | the 3-point example: $D^*=p/(p+q)$, $V$, JSD, $-\\log 4$ | | 4 | "D is logistic regression" | a logistic regression trained on $\\mathcal N(0,1)$ vs $\\mathcal N(2,1)$ recovers $D^*(x)=\\sigma(2-2x)$ | | 5 | saturating vs non-saturating | the gradient table, then the same GAN trained both ways from a strong $D$ | | 6 | full GAN | 1-D bimodal data, snapshots; $D$'s loss settles near $\\log 4$ | | 7 | mode collapse | same code, 3 modes: a healthy seed vs a collapsed seed | Total run time: about 1 minute on a laptop CPU.""") code("""import math, time, numpy as np, torch, torch.nn as nn, matplotlib.pyplot as plt, os torch.set_num_threads(4) def get_device(): if torch.cuda.is_available(): return torch.device('cuda') if torch.backends.mps.is_available(): return torch.device('mps') return torch.device('cpu') print('best available device:', get_device()) DEVICE = torch.device('cpu') # these models are tiny: the CPU is faster than moving batches to a GPU FIG = 'figures_0929'; os.makedirs(FIG, exist_ok=True) bce = nn.BCEWithLogitsLoss() # sigmoid + binary cross-entropy = the logistic-regression loss def mlp(i, h, o): # the Week 9 recipe: Linear + nonlinearity, stacked return nn.Sequential(nn.Linear(i, h), nn.LeakyReLU(0.2), nn.Linear(h, h), nn.LeakyReLU(0.2), nn.Linear(h, o))""") md("""## 2. A generator is just a sampler: $x = G(z)$, $z\\sim\\mathcal N(0,1)$ A linear $G(z)=az+b$ can only produce $\\mathcal N(b,a^2)$ (the reparameterization $x=\\mu+\\sigma\\epsilon$ from the VAE lecture!). A nonlinear $G$ can *stretch* some regions of $z$ and *squeeze* others, and so produce any shape — e.g. two bumps.""") code("""z = torch.randn(200000) G_lin = lambda z: 1.5*z + 1.0 G_bump = lambda z: z + 2*torch.tanh(3*z) # steep near z=0: few x's land near 0 -> two bumps fig, ax = plt.subplots(1, 3, figsize=(12, 3)) for a, (x, t) in zip(ax, [(z, 'noise z ~ N(0,1)'), (G_lin(z), 'G(z) = 1.5 z + 1 -> N(1, 1.5^2)'), (G_bump(z), 'G(z) = z + 2 tanh(3z)')]): a.hist(x.numpy(), bins=200, density=True, color='C0'); a.set_title(t, fontsize=10); a.set_xlim(-6, 6) plt.tight_layout(); plt.savefig(f'{FIG}/fig_pushforward.png', dpi=130); plt.show()""") md("""## 3. The optimal discriminator on a 3-point space (the board example) $p_{data}=(0.5,0.5,0)$, $p_g=(0.25,0.25,0.5)$ on $x\\in\\{1,2,3\\}$.""") code("""p = np.array([.5, .5, 0.]); q = np.array([.25, .25, .5]) Dstar = p/(p+q) def xlogy(a, b): return np.where(a > 0, a*np.log(np.where(b > 0, b, 1)), 0.) V = xlogy(p, Dstar).sum() + xlogy(q, 1-Dstar).sum() m = (p+q)/2; KL = lambda a, b: xlogy(a, a).sum() - xlogy(a, b).sum() JSD = 0.5*KL(p, m) + 0.5*KL(q, m) print('D* =', Dstar.round(4)) print(f'V(D*, G) = {V:.4f} -log 4 = {-math.log(4):.4f} JSD = {JSD:.4f} check: -log4 + 2 JSD = {-math.log(4)+2*JSD:.4f}') q2 = p.copy(); print('if p_g = p_data: D* =', (p/(p+q2+1e-300)).round(3)[:2], ' V =', round(xlogy(p, np.full(3, .5)).sum()*2, 4))""") md("""## 4. "The discriminator is a logistic regression" Real $\\sim\\mathcal N(0,1)$, fake $\\sim\\mathcal N(2,1)$. Theory: $D^*(x)=\\dfrac{p(x)}{p(x)+q(x)}=\\sigma(2-2x)$ — linear log-odds, exactly the equal-variance GDA/LDA result from HW4. A 1-parameter-pair logistic regression, trained with BCE, should find weight $\\approx-2$, bias $\\approx 2$.""") code("""torch.manual_seed(0) D = nn.Linear(1, 1) # logistic regression: logit = w x + b opt = torch.optim.Adam(D.parameters(), lr=0.05) for t in range(1500): real, fake = torch.randn(512, 1), 2 + torch.randn(512, 1) loss = bce(D(real), torch.ones(512, 1)) + bce(D(fake), torch.zeros(512, 1)) opt.zero_grad(); loss.backward(); opt.step() print(f'learned w = {D.weight.item():.3f}, b = {D.bias.item():.3f} (theory: -2, 2); final D loss {loss.item():.3f}') xs = torch.linspace(-3, 5, 400)[:, None] dens = lambda x, mu: torch.exp(-(x-mu)**2/2)/math.sqrt(2*math.pi) fig, ax = plt.subplots(figsize=(6.5, 3.2)) ax.plot(xs, dens(xs, 0), 'C0', label='real p(x) = N(0,1)'); ax.plot(xs, dens(xs, 2), 'C3', label='fake q(x) = N(2,1)') ax.plot(xs, torch.sigmoid(2-2*xs), 'k--', label='D*(x) = p/(p+q)'); ax.plot(xs, torch.sigmoid(D(xs)).detach(), 'C2', alpha=.7, lw=3, label='learned logistic regression') ax.legend(fontsize=8); ax.set_xlabel('x'); plt.tight_layout(); plt.savefig(f'{FIG}/fig_dstar.png', dpi=130); plt.show()""") md("""## 5. Saturating vs. non-saturating generator loss With $s$ = D's logit on a fake and $D=\\sigma(s)$: saturating loss $\\log(1-\\sigma(s))$ has $|\\partial/\\partial s|=D$; non-saturating $-\\log\\sigma(s)$ has $|\\partial/\\partial s| = 1-D$.""") code("""print(' D(G(z)) |grad| saturating |grad| non-saturating ratio') for Dv in [.001, .01, .1, .5, .9]: print(f'{Dv:8.3f} {Dv:12.3f} {1-Dv:12.3f} {(1-Dv)/Dv:7.1f}')""") code("""def two_modes(n): return (torch.randint(0, 2, (n, 1)).float()*4 - 2) + 0.4*torch.randn(n, 1) def saturation_run(loss_type, steps=500, B=256): torch.manual_seed(0) G, D = mlp(1, 32, 1), mlp(1, 32, 1) with torch.no_grad(): G[-1].bias += 6.0 # G starts far from the data (near x = +6) ... oD = torch.optim.Adam(D.parameters(), lr=1e-3) for t in range(300): # ... and D gets a head start: it rejects fakes confidently lD = bce(D(two_modes(B)), torch.ones(B, 1)) + bce(D(G(torch.randn(B, 1)).detach()), torch.zeros(B, 1)) oD.zero_grad(); lD.backward(); oD.step() oG, oD = torch.optim.SGD(G.parameters(), lr=0.05), torch.optim.SGD(D.parameters(), lr=0.05) # plain SGD: gradient size matters log = [] for t in range(steps): lD = bce(D(two_modes(B)), torch.ones(B, 1)) + bce(D(G(torch.randn(B, 1)).detach()), torch.zeros(B, 1)) oD.zero_grad(); lD.backward(); oD.step() s = D(G(torch.randn(B, 1))) lG = bce(s, torch.ones(B, 1)) if loss_type == 'non-saturating' else -bce(s, torch.zeros(B, 1)) oG.zero_grad(); lG.backward() gnorm = torch.sqrt(sum((p.grad**2).sum() for p in G.parameters())).item(); oG.step() log.append((torch.sigmoid(s).mean().item(), gnorm, G(torch.randn(2000, 1)).mean().item())) return np.array(log) fig, ax = plt.subplots(1, 2, figsize=(11, 3.2)) for lt in ['saturating', 'non-saturating']: L = saturation_run(lt) print(f'{lt:15s} step 0: D(G(z)) = {L[0,0]:.4f}, |grad G| = {L[0,1]:.3f}; step 100: D(G(z)) = {L[100,0]:.3f}, mean G(z) = {L[100,2]:.2f}') ax[0].plot(L[:, 0], label=lt); ax[1].plot(L[:, 2], label=lt) ax[0].set_title('mean D(G(z)) (0.5 = D fooled half the time)'); ax[1].set_title('mean of generated x (data mean = 0)') for a in ax: a.set_xlabel('G step'); a.legend() plt.tight_layout(); plt.savefig(f'{FIG}/fig_saturation.png', dpi=130); plt.show()""") md("""**Remark.** With the Adam optimizer the difference is much smaller, because Adam divides each gradient by its own running size — one reason practical GANs (and HW5) use Adam *and* the non-saturating loss. ## 6. A full 1-D GAN on two bumps Alternate: one D step (real → 1, fake → 0, fakes **detached**), one G step (non-saturating: fakes → 1, D frozen).""") code("""def train_gan(sampler, seed, steps=3000, lr=1e-3, B=256, snaps=(0, 200, 1000, 3000)): torch.manual_seed(seed) G, D = mlp(1, 32, 1), mlp(1, 32, 1) oG = torch.optim.Adam(G.parameters(), lr=lr, betas=(0.5, 0.999)); oD = torch.optim.Adam(D.parameters(), lr=lr, betas=(0.5, 0.999)) shots, dloss = {}, [] for t in range(steps + 1): if t in snaps: shots[t] = G(torch.randn(20000, 1)).detach().numpy().ravel() if t == steps: break x, fake = sampler(B), G(torch.randn(B, 1)).detach() lD = bce(D(x), torch.ones(B, 1)) + bce(D(fake), torch.zeros(B, 1)) oD.zero_grad(); lD.backward(); oD.step() lG = bce(D(G(torch.randn(B, 1))), torch.ones(B, 1)) oG.zero_grad(); lG.backward(); oG.step() dloss.append(lD.item()) return shots, np.array(dloss) t0 = time.time(); shots, dloss = train_gan(two_modes, seed=0) real = two_modes(20000).numpy().ravel() fig, ax = plt.subplots(1, 5, figsize=(15, 2.8)) for a, (t, g) in zip(ax, shots.items()): a.hist(real, bins=120, density=True, alpha=.4, color='gray', label='real'); a.hist(g, bins=120, density=True, alpha=.6, color='C1', label='G(z)') a.set_xlim(-5, 5); a.set_title(f'step {t}'); a.set_yticks([]) ax[0].legend(fontsize=8) ax[4].plot(np.convolve(dloss, np.ones(50)/50, 'valid')); ax[4].axhline(math.log(4), color='k', ls='--', label='log 4 = 1.386') ax[4].set_title('D loss (real + fake BCE)'); ax[4].legend(fontsize=8) plt.tight_layout(); plt.savefig(f'{FIG}/fig_gan_snapshots.png', dpi=130); plt.show() print(f'fraction of G(z) < 0: {(shots[3000] < 0).mean():.2f} (data: 0.50); mean D loss over last 500 steps: {dloss[-500:].mean():.3f}; {time.time()-t0:.0f}s')""") md("""## 7. Mode collapse: same code, three bumps, two seeds Nothing in the GAN objective rewards *diversity* — only fooling the current D. Some runs cover all modes; others settle on a subset. (Which seed collapses can differ across machines/library versions; the phenomenon doesn't.)""") code("""def three_modes(n): return (torch.randint(0, 3, (n, 1)).float() - 1)*4 + 0.25*torch.randn(n, 1) real3 = three_modes(20000).numpy().ravel() fig, ax = plt.subplots(1, 2, figsize=(10, 2.8)) for a, seed in zip(ax, [0, 2]): shots, _ = train_gan(three_modes, seed=seed, snaps=(3000,)) g = shots[3000]; frac = [((g > lo) & (g < hi)).mean() for lo, hi in [(-6, -2), (-2, 2), (2, 6)]] a.hist(real3, bins=150, density=True, alpha=.4, color='gray', label='real'); a.hist(g, bins=150, density=True, alpha=.6, color='C1', label='G(z)') a.set_xlim(-6.5, 6.5); a.set_yticks([]); a.set_title(f'seed {seed}: mass per mode = {np.round(frac, 2)}', fontsize=10); a.legend(fontsize=8) plt.tight_layout(); plt.savefig(f'{FIG}/fig_mode_collapse.png', dpi=130); plt.show()""") nb['cells'] = C nb.metadata['kernelspec'] = {"name": "python3", "display_name": "Python 3", "language": "python"} nbf.write(nb, '/Users/vip/Dropbox/CS6140/3_generative_models/lecture_notes/GAN/gan_lecture_demo.ipynb')