import%20marimo%0A%0A__generated_with%20%3D%20%220.25.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%0A%20%20%20%20return%20(mo%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20METADATA%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22id%22%3A%20%22loss_optimization%22%2C%0A%20%20%20%20%20%20%20%20%22name%22%3A%20%22Loss%20and%20Optimization%22%2C%0A%20%20%20%20%20%20%20%20%22kind%22%3A%20%22concept%22%2C%0A%20%20%20%20%20%20%20%20%22difficulty%22%3A%20%22intermediate%22%2C%0A%20%20%20%20%20%20%20%20%22status%22%3A%20%22complete%22%2C%0A%20%20%20%20%20%20%20%20%22topics%22%3A%20%5B%22loss_functions%22%2C%20%22optimization%22%5D%2C%0A%20%20%20%20%20%20%20%20%22related_models%22%3A%20%5B%5D%2C%0A%20%20%20%20%20%20%20%20%22prerequisites%22%3A%20%5B%22train_validation_test%22%5D%2C%0A%20%20%20%20%7D%0A%20%20%20%20mo.md(f%22%23%20%7BMETADATA%5B'name'%5D%7D%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Two%20different%20questions%0A%0A%20%20%20%20The%20loss%20defines%20what%20counts%20as%20a%20training%20mistake%3B%20the%20optimizer%20searches%20for%20parameters%20that%20reduce%20that%20loss.%0A%0A%20%20%20%20%23%23%20The%20landscape%20and%20the%20movement%0A%0A%20%20%20%20The%20loss%20draws%20the%20landscape%20and%20the%20optimizer%20chooses%20the%20walking%20strategy.%20A%20perfect%20walker%20still%20reaches%20the%20wrong%20destination%20when%20the%20landscape%20encodes%20the%20wrong%20objective.%0A%0A%20%20%20%20%23%23%20Why%20the%20distinction%20matters%0A%0A%20%20%20%20Squared%20error%20emphasizes%20large%20residuals%2C%20absolute%20error%20is%20more%20robust%2C%20and%20cross-entropy%20shapes%20probabilities.%20Optimization%20settings%20then%20determine%20whether%20training%20can%20reach%20a%20useful%20solution.%0A%0A%20%20%20%20%23%23%20What%20we%20will%20compare%0A%0A%20%20%20%20**Question%3A**%20how%20do%20loss%20shape%20and%20learning%20rate%20change%20the%20behavior%20of%20a%20simple%20fitted%20line%3F%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Loss%20and%20optimizer%20are%20different%20objects%0A%0A%20%20%20%20Let%20%24%5Ctheta%24%20represent%20all%20adjustable%20parameters.%20A%20loss%20function%20gives%20one%20number%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cmathcal%20L(%5Ctheta)%3D%5Ctext%7Berror%20produced%20by%20the%20current%20parameters%7D.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20loss%20defines%20the%20goal.%20The%20optimizer%20defines%20how%20to%20search%20for%20lower%20values%20of%20that%20goal.%0A%0A%20%20%20%20Changing%20the%20optimizer%20does%20not%20change%20what%20counts%20as%20an%20error.%20Changing%20the%20loss%20does.%0A%0A%20%20%20%20%23%23%20The%20gradient%20gives%20a%20local%20direction%0A%0A%20%20%20%20For%20one%20parameter%20%24w%24%2C%20the%20derivative%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cfrac%7Bd%5Cmathcal%20L%7D%7Bdw%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20describes%20how%20the%20loss%20changes%20after%20a%20small%20increase%20in%20%24w%24.%0A%0A%20%20%20%20-%20If%20%24d%5Cmathcal%20L%2Fdw%3E0%24%2C%20increasing%20%24w%24%20raises%20the%20loss%2C%20so%20move%20%24w%24%20downward.%0A%20%20%20%20-%20If%20%24d%5Cmathcal%20L%2Fdw%3C0%24%2C%20increasing%20%24w%24%20lowers%20the%20loss%2C%20so%20move%20%24w%24%20upward.%0A%20%20%20%20-%20If%20%24d%5Cmathcal%20L%2Fdw%3D0%24%2C%20the%20local%20slope%20is%20flat.%0A%0A%20%20%20%20For%20many%20parameters%2C%20the%20gradient%20collects%20all%20partial%20derivatives%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cnabla_%5Ctheta%5Cmathcal%20L%0A%20%20%20%20%3D%5Cbegin%7Bpmatrix%7D%0A%20%20%20%20%5Cpartial%5Cmathcal%20L%2F%5Cpartial%5Ctheta_1%5C%5C%0A%20%20%20%20%5Cvdots%5C%5C%0A%20%20%20%20%5Cpartial%5Cmathcal%20L%2F%5Cpartial%5Ctheta_m%0A%20%20%20%20%5Cend%7Bpmatrix%7D.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%23%23%20Gradient%20descent%0A%0A%20%20%20%20The%20basic%20update%20is%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Ctheta_%7Bt%2B1%7D%0A%20%20%20%20%3D%5Ctheta_t-%5Ceta%5Cnabla_%5Ctheta%5Cmathcal%20L(%5Ctheta_t)%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%24%5Ceta%3E0%24%20is%20the%20learning%20rate.%20The%20minus%20sign%20moves%20against%20the%20direction%20of%20increasing%20loss.%0A%0A%20%20%20%20%23%23%23%20A%20complete%20one-parameter%20step%0A%0A%20%20%20%20Let%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cmathcal%20L(w)%3D(w-3)%5E2.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20minimum%20is%20at%20%24w%3D3%24.%20Differentiate%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cfrac%7Bd%5Cmathcal%20L%7D%7Bdw%7D%3D2(w-3).%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Start%20with%20%24w_0%3D0%24%20and%20%24%5Ceta%3D0.1%24%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cfrac%7Bd%5Cmathcal%20L%7D%7Bdw%7D%5Cbigg%7C_%7Bw%3D0%7D%3D2(0-3)%3D-6.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Apply%20the%20update%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20w_1%3D0-0.1(-6)%3D0.6.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Check%20the%20loss%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cmathcal%20L(0)%3D9%2C%0A%20%20%20%20%5Cqquad%0A%20%20%20%20%5Cmathcal%20L(0.6)%3D5.76.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20One%20step%20moved%20%24w%24%20toward%20%243%24%20and%20reduced%20the%20loss.%0A%0A%20%20%20%20%23%23%20Why%20the%20learning%20rate%20matters%0A%0A%20%20%20%20-%20Too%20small%3A%20each%20step%20changes%20little%2C%20so%20learning%20is%20slow.%0A%20%20%20%20-%20Suitable%3A%20the%20loss%20falls%20steadily.%0A%20%20%20%20-%20Too%20large%3A%20steps%20can%20cross%20the%20low%20region%20repeatedly%20or%20move%20farther%20away.%0A%0A%20%20%20%20The%20gradient%20is%20local%20information.%20It%20does%20not%20promise%20that%20one%20step%20reaches%20the%20global%20minimum%2C%20or%20even%20that%20every%20problem%20has%20only%20one%20minimum.%0A%0A%20%20%20%20%23%23%20Batches%20and%20epochs%0A%0A%20%20%20%20An%20**epoch**%20is%20one%20pass%20through%20the%20training%20data.%20A%20**mini-batch**%20is%20a%20smaller%20group%20used%20for%20one%20gradient%20estimate%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cmathcal%20L_B(%5Ctheta)%3D%5Cfrac%7B1%7D%7B%7CB%7C%7D%5Csum_%7Bi%5Cin%20B%7D%5Cell_i(%5Ctheta).%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Small%20batches%20give%20noisy%20but%20frequent%20updates.%20Large%20batches%20give%20smoother%20estimates%20but%20require%20more%20computation%20per%20update.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%0A%20%20%20%20residual%20%3D%20np.linspace(-3%2C%203%2C%20300)%0A%20%20%20%20squared%20%3D%20residual%20**%202%0A%20%20%20%20absolute%20%3D%20np.abs(residual)%0A%20%20%20%20delta%20%3D%201.0%0A%20%20%20%20huber%20%3D%20np.where(np.abs(residual)%20%3C%3D%20delta%2C%200.5%20*%20residual%20**%202%2C%20delta%20*%20(np.abs(residual)%20-%200.5%20*%20delta))%0A%0A%20%20%20%20rng%20%3D%20np.random.default_rng(2)%0A%20%20%20%20x%20%3D%20np.linspace(-2%2C%202%2C%2070)%0A%20%20%20%20y%20%3D%201.4%20%2B%202.3%20*%20x%20%2B%20rng.normal(0%2C%200.5%2C%20len(x))%0A%0A%20%20%20%20def%20optimize(learning_rate%2C%20steps%3D80)%3A%0A%20%20%20%20%20%20%20%20intercept%2C%20slope%20%3D%200.0%2C%200.0%0A%20%20%20%20%20%20%20%20losses%20%3D%20%5B%5D%0A%20%20%20%20%20%20%20%20for%20_%20in%20range(steps)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20error%20%3D%20intercept%20%2B%20slope%20*%20x%20-%20y%0A%20%20%20%20%20%20%20%20%20%20%20%20losses.append(float(np.mean(error%20**%202)))%0A%20%20%20%20%20%20%20%20%20%20%20%20intercept%20-%3D%20learning_rate%20*%202%20*%20np.mean(error)%0A%20%20%20%20%20%20%20%20%20%20%20%20slope%20-%3D%20learning_rate%20*%202%20*%20np.mean(error%20*%20x)%0A%20%20%20%20%20%20%20%20return%20intercept%2C%20slope%2C%20losses%0A%0A%20%20%20%20return%20absolute%2C%20huber%2C%20optimize%2C%20plt%2C%20residual%2C%20squared%2C%20x%2C%20y%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20setup%20defines%20three%20loss%20shapes%20and%20one%20optimizer%20whose%20only%20changing%20control%20is%0A%20%20%20%20the%20learning%20rate.%20The%20next%20output%20therefore%20answers%20two%20separate%20questions%3A%20how%0A%20%20%20%20the%20losses%20value%20an%20error%2C%20and%20how%20the%20optimizer%20moves%20on%20one%20selected%20loss.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(absolute%2C%20huber%2C%20optimize%2C%20plt%2C%20residual%2C%20squared%2C%20x%2C%20y)%3A%0A%20%20%20%20rates%20%3D%20%5B0.01%2C%200.08%2C%200.7%5D%0A%20%20%20%20runs%20%3D%20%7Brate%3A%20optimize(rate)%20for%20rate%20in%20rates%7D%0A%20%20%20%20fig%2C%20axes%20%3D%20plt.subplots(1%2C%203%2C%20figsize%3D(13%2C%203.8))%0A%20%20%20%20axes%5B0%5D.plot(residual%2C%20squared%2C%20label%3D%22squared%22)%0A%20%20%20%20axes%5B0%5D.plot(residual%2C%20absolute%2C%20label%3D%22absolute%22)%0A%20%20%20%20axes%5B0%5D.plot(residual%2C%20huber%2C%20label%3D%22Huber%22)%0A%20%20%20%20axes%5B0%5D.set(title%3D%22Losses%20value%20the%20same%20residual%20differently%22%2C%20xlabel%3D%22residual%22%2C%20ylabel%3D%22loss%22%2C%20ylim%3D(0%2C%206))%0A%20%20%20%20axes%5B0%5D.legend()%0A%20%20%20%20for%20rate%2C%20(_%2C%20_%2C%20losses)%20in%20runs.items()%3A%0A%20%20%20%20%20%20%20%20axes%5B1%5D.plot(losses%2C%20label%3Df%22learning%20rate%3D%7Brate%7D%22)%0A%20%20%20%20axes%5B1%5D.set_yscale(%22log%22)%0A%20%20%20%20axes%5B1%5D.set(title%3D%22Optimization%20can%20converge%20or%20diverge%22%2C%20xlabel%3D%22step%22%2C%20ylabel%3D%22MSE%22)%0A%20%20%20%20axes%5B1%5D.legend(fontsize%3D8)%0A%20%20%20%20axes%5B2%5D.scatter(x%2C%20y%2C%20alpha%3D0.6)%0A%20%20%20%20for%20rate%2C%20(intercept%2C%20slope%2C%20_)%20in%20runs.items()%3A%0A%20%20%20%20%20%20%20%20if%20abs(intercept)%20%3C%20100%20and%20abs(slope)%20%3C%20100%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20axes%5B2%5D.plot(x%2C%20intercept%20%2B%20slope%20*%20x%2C%20label%3Df%22rate%3D%7Brate%7D%22)%0A%20%20%20%20axes%5B2%5D.set(title%3D%22Parameters%20found%20by%20gradient%20descent%22%2C%20xlabel%3D%22x%22%2C%20ylabel%3D%22target%22)%0A%20%20%20%20axes%5B2%5D.legend(fontsize%3D8)%0A%20%20%20%20plt.tight_layout()%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20A%20common%20category%20error%0A%0A%20%20%20%20Treating%20training%20loss%20as%20the%20product%20objective%2C%20or%20assuming%20a%20lower%20final%20loss%20always%20means%20better%20deployment%20behavior.%0A%0A%20%20%20%20%23%23%20Inspect%20the%20objective%20and%20the%20trajectory%20separately%0A%0A%20%20%20%20Plot%20the%20loss%20curve%2C%20gradients%2C%20and%20validation%20metrics.%20Check%20for%20divergence%2C%20plateaus%2C%20unstable%20batches%2C%20and%20disagreement%20between%20the%20optimized%20loss%20and%20the%20real%20decision%20cost.%0A%0A%20%20%20%20%23%23%20Rule%20to%20keep%0A%0A%20%20%20%20Choose%20a%20loss%20that%20represents%20the%20mistakes%20you%20care%20about%2C%20then%20tune%20optimization%20only%20to%20solve%20that%20learning%20problem%20reliably.%0A%0A%20%20%20%20%23%23%20Previous%20and%20next%0A%0A%20%20%20%20-%20Previous%3A%20%5BFrom%20Bernoulli%20to%20Log%20Loss%5D(%2Fconcepts%2Flog_loss)%20derives%20one%20specific%20objective.%0A%20%20%20%20-%20Next%3A%20%5BBackpropagation%5D(%2Fconcepts%2Fbackpropagation)%20computes%20gradients%20when%20the%20loss%20depends%20on%20several%20chained%20operations.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
0f2ad3c246830b5434b115f7e1fbb6d6