import%20marimo%0A%0A__generated_with%20%3D%20%220.25.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%0A%20%20%20%20return%20(mo%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20METADATA%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22id%22%3A%20%22log_loss%22%2C%0A%20%20%20%20%20%20%20%20%22name%22%3A%20%22Log%20Loss%2C%20Bernoulli%2C%20and%20Sigmoid%22%2C%0A%20%20%20%20%20%20%20%20%22summary%22%3A%20%22Essential%20formulas%20connecting%20Bernoulli%20labels%2C%20likelihood%2C%20sigmoid%20probabilities%2C%20and%20binary%20log%20loss.%22%2C%0A%20%20%20%20%20%20%20%20%22topics%22%3A%20%5B%22loss_functions%22%2C%20%22probability%22%2C%20%22classification%22%5D%2C%0A%20%20%20%20%7D%0A%20%20%20%20mo.md(f%22%23%20Cheat%20Sheet%20%E2%80%94%20%7BMETADATA%5B'name'%5D%7D%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20How%20to%20use%20this%20sheet%0A%0A%20%20%20%20This%20page%20is%20the%20compact%20formula%20path.%20Read%20it%20from%20left%20to%20right%20when%20you%20need%20to%0A%20%20%20%20connect%20a%20model%20score%20to%20a%20training%20loss.%20Here%20%24y%5Cin%5C%7B0%2C1%5C%7D%24%20is%20the%20observed%20label%2C%0A%20%20%20%20%24z%24%20is%20the%20unbounded%20score%2C%20and%20%24%5Chat%20p%24%20is%20the%20predicted%20probability%20of%20class%201.%0A%20%20%20%20For%20the%20reasoning%20behind%20every%20step%2C%20use%20the%0A%20%20%20%20%5Bfull%20concept%20notebook%5D(%2Fconcepts%2Flog_loss).%0A%0A%20%20%20%20%23%23%20The%20complete%20chain%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Cmathbf%7Bx%7D%0A%20%20%20%20%5Cxrightarrow%7B%5C%3Bz%3D%5Cmathbf%7Bw%7D%5E%7B%5Ctop%7D%5Cmathbf%7Bx%7D%2Bb%5C%3B%7D%0A%20%20%20%20z%0A%20%20%20%20%5Cxrightarrow%7B%5C%3B%5Chat%20p%3D%5Csigma(z)%5C%3B%7D%0A%20%20%20%20%5Chat%20p%0A%20%20%20%20%5Cxrightarrow%7B%5C%3B%5Ctext%7BBernoulli%7D%5C%3B%7D%0A%20%20%20%20P(Y%3Dy%5Cmid%5Cmathbf%7Bx%7D)%0A%20%20%20%20%5Cxrightarrow%7B%5C%3B-%5Clog%5C%3B%7D%0A%20%20%20%20L%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%23%23%20Essential%20formulas%0A%0A%20%20%20%20%7C%20Object%20%7C%20Formula%20%7C%20Meaning%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C%0A%20%20%20%20%7C%20Logit%20%7C%20%24z%3D%5Cmathbf%7Bw%7D%5E%7B%5Ctop%7D%5Cmathbf%7Bx%7D%2Bb%24%20%7C%20Unbounded%20model%20score%20%7C%0A%20%20%20%20%7C%20Sigmoid%20%7C%20%24%5Chat%20p%3D%5Csigma(z)%3D%5Cdfrac%7B1%7D%7B1%2Be%5E%7B-z%7D%7D%24%20%7C%20Probability%20assigned%20to%20class%201%20%7C%0A%20%20%20%20%7C%20Log-odds%20%7C%20%24z%3D%5Clog%5Cdfrac%7B%5Chat%20p%7D%7B1-%5Chat%20p%7D%24%20%7C%20Inverse%20relation%20between%20probability%20and%20logit%20%7C%0A%20%20%20%20%7C%20Bernoulli%20%7C%20%24P(Y%3Dy)%3Dp%5Ey(1-p)%5E%7B1-y%7D%24%20%7C%20Probability%20of%20one%20binary%20outcome%20%7C%0A%20%20%20%20%7C%20Dataset%20likelihood%20%7C%20%24%5Cmathcal%7BV%7D(%5Ctheta)%3D%5Cprod_%7Bi%3D1%7D%5En%5Chat%20p_i%5E%7By_i%7D(1-%5Chat%20p_i)%5E%7B1-y_i%7D%24%20%7C%20Probability%20assigned%20to%20all%20observed%20labels%20%7C%0A%20%20%20%20%7C%20Log-likelihood%20%7C%20%24%5Clog%5Cmathcal%7BV%7D%3D%5Csum_i%5By_i%5Clog%5Chat%20p_i%2B(1-y_i)%5Clog(1-%5Chat%20p_i)%5D%24%20%7C%20Product%20converted%20into%20a%20sum%20%7C%0A%20%20%20%20%7C%20Mean%20log%20loss%20%7C%20%24J%3D-%5Cdfrac1n%5Csum_i%5By_i%5Clog%5Chat%20p_i%2B(1-y_i)%5Clog(1-%5Chat%20p_i)%5D%24%20%7C%20Negative%20mean%20log-likelihood%20%7C%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Ctext%7Bmaximize%20likelihood%7D%0A%20%20%20%20%5Ciff%0A%20%20%20%20%5Ctext%7Bmaximize%20log-likelihood%7D%0A%20%20%20%20%5Ciff%0A%20%20%20%20%5Ctext%7Bminimize%20log%20loss%7D%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%23%23%20One%20example%0A%0A%20%20%20%20%24%24%0A%20%20%20%20L(%5Chat%20p%2Cy)%3D%0A%20%20%20%20%5Cbegin%7Bcases%7D%0A%20%20%20%20-%5Clog%5Chat%20p%2C%26y%3D1%2C%5C%5C%0A%20%20%20%20-%5Clog(1-%5Chat%20p)%2C%26y%3D0.%0A%20%20%20%20%5Cend%7Bcases%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%7C%20Target%20and%20prediction%20%7C%20Loss%20%7C%0A%20%20%20%20%7C---%7C---%3A%7C%0A%20%20%20%20%7C%20%24y%3D1%2C%5C%20%5Chat%20p%3D0.9%24%20%7C%20%24-%5Clog(0.9)%5Capprox0.105%24%20%7C%0A%20%20%20%20%7C%20%24y%3D1%2C%5C%20%5Chat%20p%3D0.1%24%20%7C%20%24-%5Clog(0.1)%5Capprox2.303%24%20%7C%0A%20%20%20%20%7C%20%24y%3D0%2C%5C%20%5Chat%20p%3D0.1%24%20%7C%20%24-%5Clog(0.9)%5Capprox0.105%24%20%7C%0A%20%20%20%20%7C%20%24y%3D0%2C%5C%20%5Chat%20p%3D0.9%24%20%7C%20%24-%5Clog(0.1)%5Capprox2.303%24%20%7C%0A%0A%20%20%20%20%23%23%20Useful%20derivatives%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cfrac%7B%5Cpartial%20L%7D%7B%5Cpartial%5Chat%20p%7D%0A%20%20%20%20%3D-%5Cfrac%7By%7D%7B%5Chat%20p%7D%2B%5Cfrac%7B1-y%7D%7B1-%5Chat%20p%7D%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20%5Cfrac%7B%5Cpartial%5Chat%20p%7D%7B%5Cpartial%20z%7D%3D%5Chat%20p(1-%5Chat%20p)%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%5Cfrac%7B%5Cpartial%20L%7D%7B%5Cpartial%20z%7D%3D%5Chat%20p-y%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20For%20the%20linear%20parameters%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Cnabla_%7B%5Cmathbf%7Bw%7D%7DL%3D(%5Chat%20p-y)%5Cmathbf%7Bx%7D%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20%5Cfrac%7B%5Cpartial%20L%7D%7B%5Cpartial%20b%7D%3D%5Chat%20p-y%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%23%23%20Stable%20form%20from%20logits%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20L(z%2Cy)%3D%5Coperatorname%7Bsoftplus%7D(z)-yz%0A%20%20%20%20%3D%5Clog(1%2Be%5Ez)-yz%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Compute%20binary%20cross-entropy%20directly%20from%20logits%20when%20possible.%20This%20avoids%20taking%20%24%5Clog(0)%24%20after%20probabilities%20have%20rounded%20to%200%20or%201.%0A%0A%20%20%20%20%23%23%20Remember%0A%0A%20%20%20%20-%20A%20**product**%20of%20observed-label%20probabilities%20gives%20the%20dataset%20likelihood.%0A%20%20%20%20-%20A%20**logarithm**%20converts%20that%20product%20into%20a%20sum.%0A%20%20%20%20-%20A%20**minus%20sign**%20converts%20maximization%20into%20a%20loss%20minimization%20problem.%0A%20%20%20%20-%20A%20**threshold**%20converts%20a%20probability%20into%20a%20decision%3B%20it%20does%20not%20define%20log%20loss.%0A%0A%20%20%20%20%23%23%20Related%20resources%0A%0A%20%20%20%20-%20%5BFull%20explanation%3A%20From%20Bernoulli%20to%20Log%20Loss%5D(%2Fconcepts%2Flog_loss)%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
12e48b4ed513bfa010cba8d5650178b0