import%20marimo%0A%0A__generated_with%20%3D%20%220.25.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%20%20%20%20import%20numpy%20as%20np%0A%0A%20%20%20%20return%20mo%2C%20np%2C%20plt%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20METADATA%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22id%22%3A%20%22activation_functions%22%2C%0A%20%20%20%20%20%20%20%20%22name%22%3A%20%22Activation%20Functions%22%2C%0A%20%20%20%20%20%20%20%20%22summary%22%3A%20%22The%20main%20activation%20functions%2C%20shown%20one%20by%20one%20with%20their%20formula%2C%20curve%2C%20and%20use%20case.%22%2C%0A%20%20%20%20%20%20%20%20%22topics%22%3A%20%5B%22neural_networks%22%2C%20%22activations%22%2C%20%22gradient%22%5D%2C%0A%20%20%20%20%7D%0A%20%20%20%20mo.md(f%22%23%20Cheat%20Sheet%20%E2%80%94%20%7BMETADATA%5B'name'%5D%7D%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20How%20to%20read%20this%20sheet%0A%0A%20%20%20%20A%20neuron%20first%20computes%20a%20**pre-activation**%3A%0A%0A%20%20%20%20%24%24z%3D%5Cmathbf%20w%5E%5Ctop%5Cmathbf%20x%2Bb.%24%24%0A%0A%20%20%20%20An%20activation%20function%20then%20computes%20%24a%3D%5Cphi(z)%24.%20Its%20output%20range%20controls%20what%0A%20%20%20%20the%20neuron%20can%20represent%2C%20while%20its%20derivative%20controls%20how%20easily%20a%20gradient%20can%0A%20%20%20%20pass%20through%20it%20during%20training.%0A%0A%20%20%20%20Use%20the%20curve%20to%20see%20the%20output%20range%2C%20the%20formula%20to%20calculate%20the%20value%2C%20and%20the%0A%20%20%20%20short%20note%20to%20decide%20where%20the%20function%20is%20normally%20used.%20Hidden-layer%20activations%0A%20%20%20%20create%20nonlinearity.%20Output-layer%20activations%20must%20also%20match%20the%20prediction%20task.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(np%2C%20plt)%3A%0A%20%20%20%20def%20activation_plot(x%2C%20y%2C%20title%2C%20color%3D%22%233A7CA5%22%2C%20y_limits%3DNone)%3A%0A%20%20%20%20%20%20%20%20paper%20%3D%20%22%23FAF7F0%22%0A%20%20%20%20%20%20%20%20ink%20%3D%20%22%2317324D%22%0A%20%20%20%20%20%20%20%20grid%20%3D%20%22%23DCE3E8%22%0A%20%20%20%20%20%20%20%20fig%2C%20ax%20%3D%20plt.subplots(figsize%3D(8%2C%203)%2C%20facecolor%3Dpaper)%0A%20%20%20%20%20%20%20%20ax.set_facecolor(paper)%0A%20%20%20%20%20%20%20%20ax.axhline(0%2C%20color%3Dink%2C%20linewidth%3D0.8)%0A%20%20%20%20%20%20%20%20ax.axvline(0%2C%20color%3Dink%2C%20linewidth%3D0.8)%0A%20%20%20%20%20%20%20%20ax.plot(x%2C%20y%2C%20color%3Dcolor%2C%20linewidth%3D2.8)%0A%20%20%20%20%20%20%20%20ax.set(xlabel%3D%22input%20x%22%2C%20ylabel%3D%22output%20f(x)%22%2C%20title%3Dtitle)%0A%20%20%20%20%20%20%20%20if%20y_limits%20is%20not%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20ax.set_ylim(*y_limits)%0A%20%20%20%20%20%20%20%20ax.grid(color%3Dgrid%2C%20linewidth%3D0.7)%0A%20%20%20%20%20%20%20%20ax.spines%5B%5B%22top%22%2C%20%22right%22%5D%5D.set_visible(False)%0A%20%20%20%20%20%20%20%20fig.tight_layout()%0A%20%20%20%20%20%20%20%20return%20fig%0A%0A%20%20%20%20x%20%3D%20np.linspace(-6%2C%206%2C%20500)%0A%20%20%20%20sigmoid%20%3D%201%20%2F%20(1%20%2B%20np.exp(-x))%0A%20%20%20%20return%20activation_plot%2C%20sigmoid%2C%20x%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Identity%0A%0A%20%20%20%20%24%24f(x)%3Dx%24%24%0A%0A%20%20%20%20The%20output%20is%20exactly%20equal%20to%20the%20input.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(activation_plot%2C%20x)%3A%0A%20%20%20%20activation_plot(x%2C%20x%2C%20%22Identity%22%2C%20%22%233A7CA5%22%2C%20(-6%2C%206))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(%22%22%22%0A%20%20%20%20**Why%20use%20it%20in%20deep%20learning%3F**%20For%20the%20output%20layer%20of%20unbounded%20regression%3A%20the%20prediction%20can%20take%20any%20real%20value.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Sigmoid%0A%0A%20%20%20%20%24%24f(x)%3D%5Csigma(x)%3D%5Cfrac%7B1%7D%7B1%2Be%5E%7B-x%7D%7D%24%24%0A%0A%20%20%20%20Squashes%20every%20value%20between%20%240%24%20and%20%241%24%2C%20with%20an%20S-shaped%20curve.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(activation_plot%2C%20sigmoid%2C%20x)%3A%0A%20%20%20%20activation_plot(x%2C%20sigmoid%2C%20%22Sigmoid%22%2C%20%22%23E76F51%22%2C%20(-0.1%2C%201.1))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(%22%22%22%0A%20%20%20%20**Why%20use%20it%20in%20deep%20learning%3F**%20To%20turn%20a%20logit%20into%20a%20probability%20between%200%20and%201%2C%20especially%20for%20binary%20classification%20and%20LSTM%2FGRU%20gates.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Tanh%0A%0A%20%20%20%20%24%24f(x)%3D%5Ctanh(x)%3D%5Cfrac%7Be%5Ex-e%5E%7B-x%7D%7D%7Be%5Ex%2Be%5E%7B-x%7D%7D%24%24%0A%0A%20%20%20%20Squashes%20every%20value%20between%20%24-1%24%20and%20%241%24%20and%20stays%20centered%20around%20zero.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(activation_plot%2C%20np%2C%20x)%3A%0A%20%20%20%20activation_plot(x%2C%20np.tanh(x)%2C%20%22Tanh%22%2C%20%22%232A9D8F%22%2C%20(-1.2%2C%201.2))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(%22%22%22%0A%20%20%20%20**Why%20use%20it%20in%20deep%20learning%3F**%20To%20produce%20zero-centered%20values%20bounded%20between%20%E2%88%921%20and%201%2C%20especially%20in%20recurrent%20networks.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20ReLU%0A%0A%20%20%20%20%24%24f(x)%3D%5Cmax(0%2Cx)%24%24%0A%0A%20%20%20%20Sets%20negative%20values%20to%20zero%20and%20keeps%20positive%20values%20unchanged.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(activation_plot%2C%20np%2C%20x)%3A%0A%20%20%20%20activation_plot(x%2C%20np.maximum(0%2C%20x)%2C%20%22ReLU%22%2C%20%22%233A7CA5%22%2C%20(-1%2C%206))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(%22%22%22%0A%20%20%20%20**Why%20use%20it%20in%20deep%20learning%3F**%20To%20add%20a%20fast%2C%20simple%20non-linearity%20while%20keeping%20a%20strong%20positive-side%20gradient%3B%20it%20is%20a%20common%20default%20for%20hidden%20layers.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Leaky%20ReLU%0A%0A%20%20%20%20%24%24f(x)%3D%5Cbegin%7Bcases%7Dx%2C%26x%5Cge0%5C%5C%20%5Calpha%20x%2C%26x%3C0%5Cend%7Bcases%7D%24%24%0A%0A%20%20%20%20Works%20like%20ReLU%20but%20keeps%20a%20small%20negative%20slope.%20Usually%20%24%5Calpha%3D0.01%24.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(activation_plot%2C%20np%2C%20x)%3A%0A%20%20%20%20activation_plot(x%2C%20np.where(x%20%3E%3D%200%2C%20x%2C%200.01%20*%20x)%2C%20%22Leaky%20ReLU%20%E2%80%94%20%CE%B1%20%3D%200.01%22%2C%20%22%237C5C9E%22%2C%20(-1%2C%206))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(%22%22%22%0A%20%20%20%20**Why%20use%20it%20in%20deep%20learning%3F**%20To%20keep%20a%20small%20gradient%20for%20negative%20inputs%20and%20reduce%20the%20risk%20of%20permanently%20inactive%20ReLU%20neurons.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20ELU%0A%0A%20%20%20%20%24%24f(x)%3D%5Cbegin%7Bcases%7Dx%2C%26x%3E0%5C%5C%20%5Calpha(e%5Ex-1)%2C%26x%5Cle0%5Cend%7Bcases%7D%24%24%0A%0A%20%20%20%20Stays%20linear%20on%20the%20positive%20side%20and%20becomes%20smooth%20and%20negative%20on%20the%20left.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(activation_plot%2C%20np%2C%20x)%3A%0A%20%20%20%20activation_plot(x%2C%20np.where(x%20%3E%200%2C%20x%2C%20np.exp(x)%20-%201)%2C%20%22ELU%20%E2%80%94%20%CE%B1%20%3D%201%22%2C%20%22%23E9A23B%22%2C%20(-1.5%2C%206))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(%22%22%22%0A%20%20%20%20**Why%20use%20it%20in%20deep%20learning%3F**%20To%20preserve%20negative%20outputs%20and%20a%20smooth%20transition%2C%20which%20can%20move%20the%20mean%20activation%20closer%20to%20zero.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20SELU%0A%0A%20%20%20%20%24%24f(x)%3D%5Clambda%5Cbegin%7Bcases%7Dx%2C%26x%3E0%5C%5C%20%5Calpha(e%5Ex-1)%2C%26x%5Cle0%5Cend%7Bcases%7D%24%24%0A%0A%20%20%20%20A%20scaled%20version%20of%20ELU%2C%20with%20%24%5Calpha%5Capprox1.6733%24%20and%20%24%5Clambda%5Capprox1.0507%24.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(activation_plot%2C%20np%2C%20x)%3A%0A%20%20%20%20_alpha%20%3D%201.67326324%0A%20%20%20%20_scale%20%3D%201.05070098%0A%20%20%20%20_y%20%3D%20_scale%20*%20np.where(x%20%3E%200%2C%20x%2C%20_alpha%20*%20(np.exp(x)%20-%201))%0A%20%20%20%20activation_plot(x%2C%20_y%2C%20%22SELU%22%2C%20%22%23C56B45%22%2C%20(-2%2C%206.5))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(%22%22%22%0A%20%20%20%20**Why%20use%20it%20in%20deep%20learning%3F**%20To%20build%20self-normalizing%20networks%20whose%20activations%20tend%20to%20keep%20a%20mean%20near%200%20and%20variance%20near%201.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Softplus%0A%0A%20%20%20%20%24%24f(x)%3D%5Cln(1%2Be%5Ex)%24%24%0A%0A%20%20%20%20A%20smooth%20version%20of%20ReLU%20whose%20output%20is%20always%20positive.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(activation_plot%2C%20np%2C%20x)%3A%0A%20%20%20%20activation_plot(x%2C%20np.logaddexp(0%2C%20x)%2C%20%22Softplus%22%2C%20%22%232A9D8F%22%2C%20(-1%2C%206))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(%22%22%22%0A%20%20%20%20**Why%20use%20it%20in%20deep%20learning%3F**%20To%20obtain%20a%20positive%20output%20with%20a%20smooth%20derivative%2C%20such%20as%20a%20variance%2C%20scale%2C%20or%20rate.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20SiLU%20%2F%20Swish%0A%0A%20%20%20%20%24%24f(x)%3Dx%5Csigma(x)%24%24%0A%0A%20%20%20%20A%20smooth%20activation%20that%20preserves%20a%20small%20part%20of%20negative%20values.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(activation_plot%2C%20sigmoid%2C%20x)%3A%0A%20%20%20%20activation_plot(x%2C%20x%20*%20sigmoid%2C%20%22SiLU%20%2F%20Swish%22%2C%20%22%237C5C9E%22%2C%20(-1%2C%206))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(%22%22%22%0A%20%20%20%20**Why%20use%20it%20in%20deep%20learning%3F**%20To%20provide%20a%20smooth%20activation%20with%20a%20continuous%20gradient%3B%20it%20is%20common%20in%20modern%20vision%20and%20diffusion%20architectures.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20GELU%0A%0A%20%20%20%20%24%24f(x)%3Dx%5CPhi(x)%24%24%0A%0A%20%20%20%20Gradually%20suppresses%20small%20negative%20values%20instead%20of%20cutting%20them%20off%20abruptly.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(activation_plot%2C%20np%2C%20x)%3A%0A%20%20%20%20_gelu%20%3D%200.5%20*%20x%20*%20(%0A%20%20%20%20%20%20%20%201%20%2B%20np.tanh(np.sqrt(2%20%2F%20np.pi)%20*%20(x%20%2B%200.044715%20*%20x**3))%0A%20%20%20%20)%0A%20%20%20%20activation_plot(x%2C%20_gelu%2C%20%22GELU%22%2C%20%22%23E76F51%22%2C%20(-1%2C%206))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20**Why%20use%20it%20in%20deep%20learning%3F**%20To%20gradually%20weight%20small%20inputs%20instead%20of%20cutting%20them%20off%3B%20it%20is%20the%20standard%20activation%20in%20many%20Transformers.%20Here%2C%20%24%5CPhi%24%20is%20the%20normal%20cumulative%20distribution%20function.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Mish%0A%0A%20%20%20%20%24%24f(x)%3Dx%5Ctanh(%5Cln(1%2Be%5Ex))%24%24%0A%0A%20%20%20%20A%20smooth%2C%20non-monotonic%20activation%20that%20preserves%20a%20small%20negative%20signal.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(activation_plot%2C%20np%2C%20x)%3A%0A%20%20%20%20_mish%20%3D%20x%20*%20np.tanh(np.logaddexp(0%2C%20x))%0A%20%20%20%20activation_plot(x%2C%20_mish%2C%20%22Mish%22%2C%20%22%23C56B45%22%2C%20(-1%2C%206))%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(%22%22%22%0A%20%20%20%20**Why%20use%20it%20in%20deep%20learning%3F**%20To%20use%20a%20smooth%20activation%20that%20keeps%20a%20small%20negative%20signal%3B%20it%20can%20replace%20ReLU%20in%20some%20vision%20networks.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Softmax%0A%0A%20%20%20%20%24%24f(x_i)%3D%5Cfrac%7Be%5E%7Bx_i%7D%7D%7B%5Csum_j%20e%5E%7Bx_j%7D%7D%24%24%0A%0A%20%20%20%20Turns%20a%20vector%20of%20logits%20into%20positive%20probabilities%20that%20sum%20to%20%241%24.%0A%0A%20%20%20%20**Example%3A**%0A%0A%20%20%20%20%24%24%5Coperatorname%7Bsoftmax%7D(1%2C2%2C3)%5Capprox(0.09%2C0.24%2C0.67)%24%24%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(np%2C%20plt)%3A%0A%20%20%20%20_labels%20%3D%20%5B%22class%201%22%2C%20%22class%202%22%2C%20%22class%203%22%5D%0A%20%20%20%20_logits%20%3D%20np.array(%5B1.0%2C%202.0%2C%203.0%5D)%0A%20%20%20%20_probabilities%20%3D%20np.exp(_logits%20-%20np.max(_logits))%0A%20%20%20%20_probabilities%20%2F%3D%20_probabilities.sum()%0A%20%20%20%20_fig%2C%20_ax%20%3D%20plt.subplots(figsize%3D(8%2C%203)%2C%20facecolor%3D%22%23FAF7F0%22)%0A%20%20%20%20_ax.set_facecolor(%22%23FAF7F0%22)%0A%20%20%20%20_bars%20%3D%20_ax.bar(_labels%2C%20_probabilities%2C%20color%3D%5B%22%233A7CA5%22%2C%20%22%232A9D8F%22%2C%20%22%23E76F51%22%5D)%0A%20%20%20%20_ax.bar_label(_bars%2C%20labels%3D%5Bf%22%7B_p%3A.2f%7D%22%20for%20_p%20in%20_probabilities%5D%2C%20padding%3D3)%0A%20%20%20%20_ax.set(ylabel%3D%22probability%22%2C%20title%3D%22Softmax%20of%20logits%20(1%2C%202%2C%203)%22%2C%20ylim%3D(0%2C%200.8))%0A%20%20%20%20_ax.spines%5B%5B%22top%22%2C%20%22right%22%5D%5D.set_visible(False)%0A%20%20%20%20_ax.grid(axis%3D%22y%22%2C%20color%3D%22%23DCE3E8%22%2C%20linewidth%3D0.7)%0A%20%20%20%20_fig.tight_layout()%0A%20%20%20%20_fig%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(%22%22%22%0A%20%20%20%20**Why%20use%20it%20in%20deep%20learning%3F**%20To%20turn%20an%20output%20layer's%20logits%20into%20probabilities%20that%20sum%20to%201%20for%20multiclass%20classification.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Quick%20choice%0A%0A%20%20%20%20%7C%20Need%20%7C%20Activation%20%7C%0A%20%20%20%20%7C---%7C---%7C%0A%20%20%20%20%7C%20standard%20hidden%20layers%20%7C%20ReLU%20%7C%0A%20%20%20%20%7C%20avoid%20dead%20neurons%20%7C%20Leaky%20ReLU%20%7C%0A%20%20%20%20%7C%20Transformer%20%7C%20GELU%20or%20SiLU%20%7C%0A%20%20%20%20%7C%20regression%20output%20%7C%20identity%20%7C%0A%20%20%20%20%7C%20positive%20output%20%7C%20Softplus%20%7C%0A%20%20%20%20%7C%20binary%20classification%20%7C%20sigmoid%20%7C%0A%20%20%20%20%7C%20multiclass%20classification%20%7C%20softmax%20%7C%0A%20%20%20%20%7C%20output%20between%20%24-1%24%20and%20%241%24%20%7C%20tanh%20%7C%0A%0A%20%20%20%20%23%23%20Related%20resources%0A%0A%20%20%20%20-%20%5BThe%20Perceptron%5D(%2Fconcepts%2Fperceptron)%20%E2%80%94%20understand%20a%20hard%20threshold%20as%20a%20decision%20activation.%0A%20%20%20%20-%20%5BFrom%20Bernoulli%20to%20Log%20Loss%5D(%2Fconcepts%2Flog_loss)%20%E2%80%94%20connect%20sigmoid%2C%20probability%2C%20and%20binary%20loss.%0A%20%20%20%20-%20%5BDeep%20Learning%20Essential%20Formulas%5D(%2Fcheatsheets%2Fdeep_learning)%20%E2%80%94%20place%20activations%20inside%20the%20complete%20training%20loop.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
ddffd6cb6a8f94c0259cad7907897816