import%20marimo%0A%0A__generated_with%20%3D%20%220.25.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%0A%20%20%20%20return%20(mo%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20METADATA%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22id%22%3A%20%22deep_learning%22%2C%0A%20%20%20%20%20%20%20%20%22name%22%3A%20%22Deep%20Learning%20Essential%20Formulas%22%2C%0A%20%20%20%20%20%20%20%20%22summary%22%3A%20%22A%20compact%20map%20of%20neural-network%20formulas%20from%20the%20forward%20pass%20through%20backpropagation%2C%20optimization%2C%20convolution%2C%20and%20attention.%22%2C%0A%20%20%20%20%20%20%20%20%22topics%22%3A%20%5B%22deep_learning%22%2C%20%22neural_networks%22%2C%20%22backpropagation%22%2C%20%22optimization%22%5D%2C%0A%20%20%20%20%7D%0A%20%20%20%20mo.md(f%22%23%20Cheat%20Sheet%20%E2%80%94%20%7BMETADATA%5B'name'%5D%7D%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20How%20to%20use%20this%20sheet%0A%0A%20%20%20%20This%20is%20a%20map%20of%20the%20main%20equations%2C%20not%20a%20replacement%20for%20their%20explanations.%0A%20%20%20%20Follow%20the%20training%20path%20from%20left%20to%20right%20to%20locate%20a%20formula.%20Then%20use%20the%20linked%0A%20%20%20%20concept%20notebooks%20at%20the%20end%20for%20the%20derivation%2C%20intuition%2C%20and%20worked%20examples.%0A%0A%20%20%20%20%23%23%20The%20training%20path%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Ctext%7Binput%20%7D%5Cmathbf%7Bx%7D%0A%20%20%20%20%5Cxrightarrow%7B%5Ctext%7Blayers%7D%7D%0A%20%20%20%20%5Ctext%7Blogits%20%7D%5Cmathbf%7Bz%7D%0A%20%20%20%20%5Cxrightarrow%7B%5Ctext%7Boutput%20function%7D%7D%0A%20%20%20%20%5Chat%7B%5Cmathbf%20y%7D%0A%20%20%20%20%5Cxrightarrow%7B%5Ctext%7Bloss%7D%7D%0A%20%20%20%20%5Cmathcal%20L%0A%20%20%20%20%5Cxrightarrow%7B%5Ctext%7Bbackpropagation%7D%7D%0A%20%20%20%20%5Cnabla_%5Ctheta%5Cmathcal%20L%0A%20%20%20%20%5Cxrightarrow%7B%5Ctext%7Boptimizer%7D%7D%0A%20%20%20%20%5Ctheta_%7Bt%2B1%7D%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20A%20neural%20network%20learns%20by%20repeating%20this%20sequence%20over%20mini-batches.%0A%0A%20%20%20%20%23%23%201.%20Neuron%20and%20dense%20layer%0A%0A%20%20%20%20One%20neuron%20computes%20a%20weighted%20sum%20followed%20by%20an%20activation%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20z%3D%5Cmathbf%20w%5E%5Ctop%5Cmathbf%20x%2Bb%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20a%3D%5Cphi(z).%0A%20%20%20%20%24%24%0A%0A%20%20%20%20A%20layer%20performs%20the%20same%20computation%20for%20several%20neurons%20at%20once%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Cmathbf%20z%5E%7B(l)%7D%3DW%5E%7B(l)%7D%5Cmathbf%20a%5E%7B(l-1)%7D%2B%5Cmathbf%20b%5E%7B(l)%7D%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20%5Cmathbf%20a%5E%7B(l)%7D%3D%5Cphi(%5Cmathbf%20z%5E%7B(l)%7D)%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20For%20%24n_%7Bin%7D%24%20inputs%20and%20%24n_%7Bout%7D%24%20outputs%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5C%23%5Ctext%7Bparameters%7D%3Dn_%7Bin%7Dn_%7Bout%7D%2Bn_%7Bout%7D.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20first%20term%20counts%20weights%3B%20the%20second%20counts%20biases.%0A%0A%20%20%20%20%23%23%202.%20Common%20activations%0A%0A%20%20%20%20%7C%20Activation%20%7C%20Formula%20%7C%20Main%20role%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C%0A%20%20%20%20%7C%20ReLU%20%7C%20%24%5Cmax(0%2Cz)%24%20%7C%20Simple%20default%20for%20many%20hidden%20layers%20%7C%0A%20%20%20%20%7C%20Leaky%20ReLU%20%7C%20%24%5Cmax(%5Calpha%20z%2Cz)%24%20%7C%20Keeps%20a%20small%20gradient%20when%20%24z%3C0%24%20%7C%0A%20%20%20%20%7C%20Sigmoid%20%7C%20%24%5Csigma(z)%3D%5Cdfrac1%7B1%2Be%5E%7B-z%7D%7D%24%20%7C%20Converts%20a%20binary%20logit%20into%20a%20probability%20%7C%0A%20%20%20%20%7C%20Tanh%20%7C%20%24%5Ctanh(z)%24%20%7C%20Produces%20an%20output%20between%20%24-1%24%20and%20%241%24%20%7C%0A%20%20%20%20%7C%20GELU%20%7C%20%24z%5CPhi(z)%24%20%7C%20Smooth%20activation%20common%20in%20Transformers%20%7C%0A%0A%20%20%20%20Without%20a%20nonlinear%20activation%20between%20layers%2C%20several%20dense%20layers%20collapse%20into%20one%20linear%20transformation.%0A%0A%20%20%20%20%23%23%203.%20Output%20layer%20and%20loss%0A%0A%20%20%20%20%23%23%23%20Regression%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Chat%20y%3Dz%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20%5Coperatorname%7BMSE%7D%3D%5Cfrac1n%5Csum_%7Bi%3D1%7D%5En(%5Chat%20y_i-y_i)%5E2.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20identity%20output%20permits%20any%20real%20value%3B%20MSE%20emphasizes%20large%20residuals.%0A%0A%20%20%20%20%23%23%23%20Binary%20classification%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Chat%20p%3D%5Csigma(z)%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20J%3D-%5Cfrac1n%5Csum_i%5By_i%5Clog%5Chat%20p_i%2B(1-y_i)%5Clog(1-%5Chat%20p_i)%5D.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Sigmoid%20combined%20with%20log%20loss%20gives%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%5Cfrac%7B%5Cpartial%20L%7D%7B%5Cpartial%20z%7D%3D%5Chat%20p-y%7D.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%23%23%23%20Multiclass%20classification%0A%0A%20%20%20%20%24%24%0A%20%20%20%20p_k%3D%5Coperatorname%7Bsoftmax%7D(%5Cmathbf%20z)_k%0A%20%20%20%20%3D%5Cfrac%7Be%5E%7Bz_k%7D%7D%7B%5Csum_j%20e%5E%7Bz_j%7D%7D%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20L%3D-%5Csum_k%20y_k%5Clog%20p_k.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Softmax%20combined%20with%20cross-entropy%20gives%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%5Cfrac%7B%5Cpartial%20L%7D%7B%5Cpartial%20z_k%7D%3Dp_k-y_k%7D.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Compute%20classification%20losses%20directly%20from%20logits%20when%20possible%20for%20numerical%20stability.%0A%0A%20%20%20%20%23%23%204.%20Backpropagation%0A%0A%20%20%20%20The%20chain%20rule%20transmits%20the%20effect%20of%20one%20variable%20through%20composed%20operations%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cfrac%7B%5Cpartial%20L%7D%7B%5Cpartial%20x%7D%0A%20%20%20%20%3D%5Cfrac%7B%5Cpartial%20L%7D%7B%5Cpartial%20y%7D%0A%20%20%20%20%5Cfrac%7B%5Cpartial%20y%7D%7B%5Cpartial%20x%7D.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20For%20a%20dense%20layer%2C%20define%20%24%5Cboldsymbol%5Cdelta%5E%7B(l)%7D%3D%5Cpartial%20L%2F%5Cpartial%5Cmathbf%20z%5E%7B(l)%7D%24.%20Then%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Cfrac%7B%5Cpartial%20L%7D%7B%5Cpartial%20W%5E%7B(l)%7D%7D%0A%20%20%20%20%3D%5Cboldsymbol%5Cdelta%5E%7B(l)%7D(%5Cmathbf%20a%5E%7B(l-1)%7D)%5E%5Ctop%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20%5Cfrac%7B%5Cpartial%20L%7D%7B%5Cpartial%5Cmathbf%20b%5E%7B(l)%7D%7D%3D%5Cboldsymbol%5Cdelta%5E%7B(l)%7D%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboldsymbol%5Cdelta%5E%7B(l-1)%7D%0A%20%20%20%20%3D%5Cleft((W%5E%7B(l)%7D)%5E%5Ctop%5Cboldsymbol%5Cdelta%5E%7B(l)%7D%5Cright)%0A%20%20%20%20%5Codot%5Cphi'(%5Cmathbf%20z%5E%7B(l-1)%7D).%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%24%5Codot%24%20denotes%20element-wise%20multiplication.%20The%20final%20equation%20propagates%20the%20error%20signal%20into%20the%20previous%20layer.%0A%0A%20%20%20%20%23%23%205.%20Parameter%20updates%0A%0A%20%20%20%20%23%23%23%20Gradient%20descent%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Ctheta_%7Bt%2B1%7D%3D%5Ctheta_t-%5Ceta%5Cnabla_%5Ctheta%5Cmathcal%20L(%5Ctheta_t)%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%24%5Ceta%24%20is%20the%20learning%20rate.%20A%20value%20that%20is%20too%20large%20can%20make%20training%20diverge%3B%20one%20that%20is%20too%20small%20makes%20progress%20slow.%0A%0A%20%20%20%20%23%23%23%20Momentum%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cmathbf%20v_t%3D%5Cbeta%5Cmathbf%20v_%7Bt-1%7D%2B%5Cmathbf%20g_t%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20%5Ctheta_%7Bt%2B1%7D%3D%5Ctheta_t-%5Ceta%5Cmathbf%20v_t.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Momentum%20smooths%20successive%20gradients%20and%20accelerates%20movement%20in%20persistent%20directions.%0A%0A%20%20%20%20%23%23%23%20Adam%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cmathbf%20m_t%3D%5Cbeta_1%5Cmathbf%20m_%7Bt-1%7D%2B(1-%5Cbeta_1)%5Cmathbf%20g_t%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20%5Cmathbf%20v_t%3D%5Cbeta_2%5Cmathbf%20v_%7Bt-1%7D%2B(1-%5Cbeta_2)%5Cmathbf%20g_t%5E2%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Chat%7B%5Cmathbf%20m%7D_t%3D%5Cfrac%7B%5Cmathbf%20m_t%7D%7B1-%5Cbeta_1%5Et%7D%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20%5Chat%7B%5Cmathbf%20v%7D_t%3D%5Cfrac%7B%5Cmathbf%20v_t%7D%7B1-%5Cbeta_2%5Et%7D%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20%5Ctheta_%7Bt%2B1%7D%3D%5Ctheta_t-%5Ceta%0A%20%20%20%20%5Cfrac%7B%5Chat%7B%5Cmathbf%20m%7D_t%7D%7B%5Csqrt%7B%5Chat%7B%5Cmathbf%20v%7D_t%7D%2B%5Cvarepsilon%7D.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Adam%20adapts%20the%20update%20scale%20for%20each%20parameter.%0A%0A%20%20%20%20%23%23%206.%20Initialization%20and%20regularization%0A%0A%20%20%20%20%7C%20Technique%20%7C%20Formula%20%7C%20Purpose%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C%0A%20%20%20%20%7C%20Xavier%20%7C%20%24%5Coperatorname%7BVar%7D(W)%5Capprox%5Cdfrac%7B2%7D%7Bn_%7Bin%7D%2Bn_%7Bout%7D%7D%24%20%7C%20Stabilizes%20activations%2C%20especially%20with%20tanh%20%7C%0A%20%20%20%20%7C%20He%20%7C%20%24%5Coperatorname%7BVar%7D(W)%5Capprox%5Cdfrac%7B2%7D%7Bn_%7Bin%7D%7D%24%20%7C%20Suits%20ReLU-like%20activations%20%7C%0A%20%20%20%20%7C%20L2%20penalty%20%7C%20%24%5Cmathcal%20L_%7Btotal%7D%3D%5Cmathcal%20L_%7Bdata%7D%2B%5Clambda%5ClVert%20W%5CrVert_2%5E2%24%20%7C%20Discourages%20large%20weights%20%7C%0A%20%20%20%20%7C%20Inverted%20dropout%20%7C%20%24%5Ctilde%7B%5Cmathbf%20h%7D%3D%5Cdfrac%7B%5Cmathbf%20m%5Codot%5Cmathbf%20h%7D%7B1-q%7D%24%2C%20%24m_j%5Csim%5Coperatorname%7BBernoulli%7D(1-q)%24%20%7C%20Randomly%20masks%20activations%20during%20training%20%7C%0A%0A%20%20%20%20Dropout%20is%20disabled%20at%20inference.%20Weight%20decay%20and%20an%20L2%20penalty%20coincide%20for%20basic%20SGD%2C%20but%20not%20necessarily%20for%20adaptive%20optimizers.%0A%0A%20%20%20%20%23%23%207.%20Normalization%0A%0A%20%20%20%20For%20a%20mini-batch%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cmu_B%3D%5Cfrac1m%5Csum_i%20x_i%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20%5Csigma_B%5E2%3D%5Cfrac1m%5Csum_i(x_i-%5Cmu_B)%5E2%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Chat%20x_i%3D%5Cfrac%7Bx_i-%5Cmu_B%7D%7B%5Csqrt%7B%5Csigma_B%5E2%2B%5Cvarepsilon%7D%7D%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20y_i%3D%5Cgamma%5Chat%20x_i%2B%5Cbeta%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Batch%20normalization%20uses%20mini-batch%20statistics.%20Layer%20normalization%20instead%20normalizes%20features%20within%20one%20example%20and%20is%20common%20in%20Transformers.%0A%0A%20%20%20%20%23%23%208.%20Convolution%0A%0A%20%20%20%20For%20a%20one-dimensional%20signal%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20y%5Bt%5D%3D%5Csum_%7Bc%3D1%7D%5E%7BC_%7Bin%7D%7D%5Csum_%7Bk%3D0%7D%5E%7BK-1%7DW%5Bc%2Ck%5Dx_c%5Bt%2Bk%5D%2Bb.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20same%20kernel%20is%20reused%20at%20every%20position%3A%20this%20is%20weight%20sharing.%20For%20a%20two-dimensional%20convolution%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5C%23%5Ctext%7Bparameters%7D%3DC_%7Bout%7D(C_%7Bin%7DK_hK_w%2B1).%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%23%23%209.%20Attention%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Coperatorname%7BAttention%7D(Q%2CK%2CV)%0A%20%20%20%20%3D%5Coperatorname%7Bsoftmax%7D%5Cleft(%5Cfrac%7BQK%5E%5Ctop%7D%7B%5Csqrt%7Bd_k%7D%7D%5Cright)V%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%24QK%5E%5Ctop%24%20measures%20compatibility%20between%20queries%20and%20keys.%20Softmax%20produces%20weights%20that%20mix%20the%20values%20%24V%24.%20Scaling%20by%20%24%5Csqrt%7Bd_k%7D%24%20prevents%20excessively%20large%20scores.%0A%0A%20%20%20%20%23%23%2010.%20Minimal%20training%20loop%0A%0A%20%20%20%201.%20Forward%20pass%3A%20compute%20logits%20and%20loss.%0A%20%20%20%202.%20Backward%20pass%3A%20compute%20gradients.%0A%20%20%20%203.%20Update%3A%20apply%20the%20optimizer.%0A%20%20%20%204.%20Repeat%20over%20mini-batches%20and%20evaluate%20separately%20on%20validation%20data.%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Ctext%7Bforward%7D%0A%20%20%20%20%5Crightarrow%0A%20%20%20%20%5Ctext%7Bloss%7D%0A%20%20%20%20%5Crightarrow%0A%20%20%20%20%5Ctext%7Bbackward%7D%0A%20%20%20%20%5Crightarrow%0A%20%20%20%20%5Ctext%7Bupdate%7D%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%23%23%20Detailed%20references%0A%0A%20%20%20%20This%20sheet%20is%20a%20map.%20Use%20the%20focused%20resources%20for%20full%20explanations%3A%0A%0A%20%20%20%20-%20%5BThe%20Perceptron%5D(%2Fconcepts%2Fperceptron)%20%E2%80%94%20neuron%2C%20score%2C%20and%20threshold.%0A%20%20%20%20-%20%5BFrom%20Bernoulli%20to%20Log%20Loss%5D(%2Fconcepts%2Flog_loss)%20%E2%80%94%20binary%20output%20and%20the%20probabilistic%20origin%20of%20its%20loss.%0A%20%20%20%20-%20%5BBackpropagation%5D(%2Fconcepts%2Fbackpropagation)%20%E2%80%94%20how%20gradients%20move%20through%20a%20network.%0A%20%20%20%20-%20%5BLoss%20and%20Optimization%5D(%2Fconcepts%2Floss_optimization)%20%E2%80%94%20objective%20functions%20and%20parameter%20updates.%0A%20%20%20%20-%20%5BActivation%20Functions%5D(%2Fcheatsheets%2Factivation_functions)%20%E2%80%94%20detailed%20formulas%20and%20curves.%0A%20%20%20%20-%20%5BLoss%20Functions%5D(%2Fcheatsheets%2Floss_functions)%20%E2%80%94%20losses%20organized%20by%20task.%0A%20%20%20%20-%20%5BDerivatives%20and%20Gradients%5D(%2Fcheatsheets%2Fderivatives)%20%E2%80%94%20calculus%20rules%20used%20by%20backpropagation.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
eba732815d75249b05668886d90c7d62