import%20marimo%0A%0A__generated_with%20%3D%20%220.25.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%0A%20%20%20%20return%20(mo%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20METADATA%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22id%22%3A%20%22regularization%22%2C%0A%20%20%20%20%20%20%20%20%22name%22%3A%20%22Regularization%22%2C%0A%20%20%20%20%20%20%20%20%22kind%22%3A%20%22concept%22%2C%0A%20%20%20%20%20%20%20%20%22difficulty%22%3A%20%22intermediate%22%2C%0A%20%20%20%20%20%20%20%20%22status%22%3A%20%22complete%22%2C%0A%20%20%20%20%20%20%20%20%22topics%22%3A%20%5B%22regularization%22%2C%20%22generalization%22%2C%20%22model_complexity%22%5D%2C%0A%20%20%20%20%20%20%20%20%22related_models%22%3A%20%5B%5D%2C%0A%20%20%20%20%20%20%20%20%22prerequisites%22%3A%20%5B%22bias_variance%22%5D%2C%0A%20%20%20%20%7D%0A%20%20%20%20mo.md(f%22%23%20%7BMETADATA%5B'name'%5D%7D%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20The%20problem%0A%0A%20%20%20%20Regularization%20makes%20some%20fitted%20solutions%20preferable%20to%20others%20so%20the%20learned%20behavior%20is%20more%20likely%20to%20survive%20new%20data.%0A%0A%20%20%20%20%23%23%20Fit%20the%20data%20while%20limiting%20freedom%0A%0A%20%20%20%20Fitting%20rewards%20agreement%20with%20training%20examples%3B%20regularization%20charges%20rent%20for%20complexity.%20The%20useful%20balance%20depends%20on%20how%20much%20signal%20and%20data%20the%20problem%20contains.%0A%0A%20%20%20%20%23%23%20Why%20a%20small%20training%20error%20is%20not%20enough%0A%0A%20%20%20%20Flexible%20models%20can%20explain%20both%20signal%20and%20noise.%20Penalties%2C%20early%20stopping%2C%20depth%20limits%2C%20dropout%2C%20and%20augmentation%20all%20constrain%20what%20details%20are%20easy%20to%20learn.%0A%0A%20%20%20%20%23%23%20What%20we%20will%20vary%0A%0A%20%20%20%20**Question%3A**%20how%20does%20an%20L2%20penalty%20stabilize%20a%20high-degree%20polynomial%20fit%3F%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Add%20a%20preference%20to%20the%20objective%0A%0A%20%20%20%20Let%20%24%5Cmathcal%20L_%7Bdata%7D(%5Cmathbf%20w)%24%20measure%20fit%20to%20the%20training%20examples.%20Regularization%20adds%20a%20term%20that%20expresses%20a%20preference%20among%20solutions%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Cmathcal%20L_%7Btotal%7D(%5Cmathbf%20w)%0A%20%20%20%20%3D%5Cmathcal%20L_%7Bdata%7D(%5Cmathbf%20w)%0A%20%20%20%20%2B%5Clambda%5C%2C%5COmega(%5Cmathbf%20w)%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%7C%20Symbol%20%7C%20Meaning%20%7C%0A%20%20%20%20%7C---%7C---%7C%0A%20%20%20%20%7C%20%24%5Cmathcal%20L_%7Bdata%7D%24%20%7C%20error%20on%20training%20examples%20%7C%0A%20%20%20%20%7C%20%24%5COmega(%5Cmathbf%20w)%24%20%7C%20penalty%20for%20an%20unwanted%20kind%20of%20complexity%20%7C%0A%20%20%20%20%7C%20%24%5Clambda%5Cgeq0%24%20%7C%20strength%20of%20the%20preference%20%7C%0A%0A%20%20%20%20When%20%24%5Clambda%3D0%24%2C%20only%20training%20fit%20matters.%20As%20%24%5Clambda%24%20grows%2C%20the%20penalty%20has%20more%20influence.%0A%0A%20%20%20%20%23%23%20L2%20regularization%0A%0A%20%20%20%20The%20L2%20penalty%20is%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5COmega(%5Cmathbf%20w)%3D%5Cfrac12%5Csum_%7Bj%3D1%7D%5E%7Bd%7Dw_j%5E2%0A%20%20%20%20%3D%5Cfrac12%5ClVert%5Cmathbf%20w%5CrVert_2%5E2.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20factor%20%241%2F2%24%20makes%20the%20derivative%20simple%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cfrac%7B%5Cpartial%7D%7B%5Cpartial%20w_j%7D%0A%20%20%20%20%5Cleft(%5Cfrac12%5Clambda%20w_j%5E2%5Cright)%0A%20%20%20%20%3D%5Clambda%20w_j.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Therefore%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Cnabla_%7B%5Cmathbf%20w%7D%5Cmathcal%20L_%7Btotal%7D%0A%20%20%20%20%3D%5Cnabla_%7B%5Cmathbf%20w%7D%5Cmathcal%20L_%7Bdata%7D%0A%20%20%20%20%2B%5Clambda%5Cmathbf%20w%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20With%20gradient%20descent%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cmathbf%20w_%7Bt%2B1%7D%0A%20%20%20%20%3D%5Cmathbf%20w_t%0A%20%20%20%20-%5Ceta%5Cleft(%0A%20%20%20%20%5Cnabla_%7B%5Cmathbf%20w%7D%5Cmathcal%20L_%7Bdata%7D%0A%20%20%20%20%2B%5Clambda%5Cmathbf%20w_t%0A%20%20%20%20%5Cright).%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Rearrange%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cmathbf%20w_%7Bt%2B1%7D%0A%20%20%20%20%3D(1-%5Ceta%5Clambda)%5Cmathbf%20w_t%0A%20%20%20%20-%5Ceta%5Cnabla_%7B%5Cmathbf%20w%7D%5Cmathcal%20L_%7Bdata%7D.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20factor%20%24(1-%5Ceta%5Clambda)%24%20pulls%20weights%20toward%20zero%20at%20every%20step.%20Smaller%20weights%20reduce%20the%20influence%20of%20individual%20features%20and%20often%20make%20the%20fitted%20rule%20less%20sensitive%20to%20small%20changes%20in%20the%20data.%0A%0A%20%20%20%20%23%23%23%20Numerical%20step%0A%0A%20%20%20%20Suppose%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20w_t%3D2%2C%5Cquad%0A%20%20%20%20%5Cfrac%7B%5Cpartial%5Cmathcal%20L_%7Bdata%7D%7D%7B%5Cpartial%20w%7D%3D-1%2C%0A%20%20%20%20%5Cquad%5Ceta%3D0.1%2C%5Cquad%5Clambda%3D0.2.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Then%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cfrac%7B%5Cpartial%5Cmathcal%20L_%7Btotal%7D%7D%7B%5Cpartial%20w%7D%0A%20%20%20%20%3D-1%2B(0.2)(2)%3D-0.6%2C%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%24%24%0A%20%20%20%20w_%7Bt%2B1%7D%3D2-0.1(-0.6)%3D2.06.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Without%20regularization%2C%20the%20update%20would%20give%20%242.1%24.%20The%20data%20still%20pushes%20the%20weight%20upward%2C%20but%20the%20penalty%20makes%20that%20movement%20smaller.%0A%0A%20%20%20%20%23%23%20L1%20regularization%0A%0A%20%20%20%20The%20L1%20penalty%20is%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5COmega(%5Cmathbf%20w)%3D%5Csum_j%7Cw_j%7C.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Away%20from%20zero%2C%20its%20derivative%20is%20%24%5Coperatorname%7Bsgn%7D(w_j)%24.%20L1%20can%20set%20some%20fitted%20weights%20exactly%20to%20zero%2C%20which%20can%20create%20sparse%20solutions.%20L2%20usually%20shrinks%20many%20weights%20without%20making%20them%20exactly%20zero.%0A%0A%20%20%20%20%23%23%20Regularization%20is%20a%20trade-off%0A%0A%20%20%20%20-%20Too%20little%20regularization%3A%20the%20fitted%20rule%20may%20follow%20unstable%20details.%0A%20%20%20%20-%20Too%20much%20regularization%3A%20the%20fitted%20rule%20may%20erase%20real%20structure.%0A%20%20%20%20-%20A%20useful%20value%20of%20%24%5Clambda%24%20is%20selected%20using%20validation%20data%2C%20never%20the%20final%20test%20set.%0A%0A%20%20%20%20Regularization%20does%20not%20mean%20%E2%80%9Cmake%20everything%20small.%E2%80%9D%20It%20means%20%E2%80%9Cstate%20which%20solutions%20should%20be%20preferred%20when%20several%20solutions%20fit%20the%20training%20data.%E2%80%9D%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%20%20%20%20from%20sklearn.linear_model%20import%20Ridge%0A%20%20%20%20from%20sklearn.metrics%20import%20mean_squared_error%0A%20%20%20%20from%20sklearn.pipeline%20import%20make_pipeline%0A%20%20%20%20from%20sklearn.preprocessing%20import%20PolynomialFeatures%2C%20StandardScaler%0A%0A%20%20%20%20rng%20%3D%20np.random.default_rng(15)%0A%20%20%20%20x_train%20%3D%20np.sort(rng.uniform(-2.5%2C%202.5%2C%2034))%0A%20%20%20%20y_train%20%3D%20np.sin(1.6%20*%20x_train)%20%2B%20rng.normal(0%2C%200.3%2C%20len(x_train))%0A%20%20%20%20grid%20%3D%20np.linspace(-2.5%2C%202.5%2C%20300)%0A%20%20%20%20truth%20%3D%20np.sin(1.6%20*%20grid)%0A%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20PolynomialFeatures%2C%0A%20%20%20%20%20%20%20%20Ridge%2C%0A%20%20%20%20%20%20%20%20StandardScaler%2C%0A%20%20%20%20%20%20%20%20grid%2C%0A%20%20%20%20%20%20%20%20make_pipeline%2C%0A%20%20%20%20%20%20%20%20mean_squared_error%2C%0A%20%20%20%20%20%20%20%20plt%2C%0A%20%20%20%20%20%20%20%20truth%2C%0A%20%20%20%20%20%20%20%20x_train%2C%0A%20%20%20%20%20%20%20%20y_train%2C%0A%20%20%20%20)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20Every%20curve%20below%20uses%20the%20same%20degree-15%20feature%20space%20and%20the%20same%20observations.%0A%20%20%20%20Only%20the%20L2%20strength%20changes.%20This%20lets%20us%20see%20the%20full%20trade-off%3A%20coefficient%20size%0A%20%20%20%20falls%20continuously%2C%20but%20new-data%20error%20improves%20only%20up%20to%20a%20point.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20PolynomialFeatures%2C%0A%20%20%20%20Ridge%2C%0A%20%20%20%20StandardScaler%2C%0A%20%20%20%20grid%2C%0A%20%20%20%20make_pipeline%2C%0A%20%20%20%20mean_squared_error%2C%0A%20%20%20%20plt%2C%0A%20%20%20%20truth%2C%0A%20%20%20%20x_train%2C%0A%20%20%20%20y_train%2C%0A)%3A%0A%20%20%20%20alphas%20%3D%20%5B0.0%2C%200.01%2C%201.0%2C%20100.0%5D%0A%20%20%20%20fig%2C%20axes%20%3D%20plt.subplots(1%2C%202%2C%20figsize%3D(11%2C%204))%0A%20%20%20%20norms%2C%20errors%20%3D%20%5B%5D%2C%20%5B%5D%0A%20%20%20%20for%20alpha%20in%20alphas%3A%0A%20%20%20%20%20%20%20%20model%20%3D%20make_pipeline(PolynomialFeatures(15)%2C%20StandardScaler()%2C%20Ridge(alpha%3Dalpha))%0A%20%20%20%20%20%20%20%20model.fit(x_train%5B%3A%2C%20None%5D%2C%20y_train)%0A%20%20%20%20%20%20%20%20prediction%20%3D%20model.predict(grid%5B%3A%2C%20None%5D)%0A%20%20%20%20%20%20%20%20ridge%20%3D%20model.named_steps%5B%22ridge%22%5D%0A%20%20%20%20%20%20%20%20norms.append(float((ridge.coef_%20**%202).sum()%20**%200.5))%0A%20%20%20%20%20%20%20%20errors.append(mean_squared_error(truth%2C%20prediction))%0A%20%20%20%20%20%20%20%20axes%5B0%5D.plot(grid%2C%20prediction%2C%20label%3Df%22alpha%3D%7Balpha%3Ag%7D%22)%0A%20%20%20%20axes%5B0%5D.scatter(x_train%2C%20y_train%2C%20color%3D%22black%22%2C%20s%3D18%2C%20alpha%3D0.6)%0A%20%20%20%20axes%5B0%5D.plot(grid%2C%20truth%2C%20color%3D%22grey%22%2C%20linestyle%3D%22--%22%2C%20label%3D%22true%20signal%22)%0A%20%20%20%20axes%5B0%5D.set(title%3D%22Penalty%20changes%20the%20preferred%20fit%22%2C%20xlabel%3D%22x%22%2C%20ylabel%3D%22target%22%2C%20ylim%3D(-2%2C%202))%0A%20%20%20%20axes%5B0%5D.legend(fontsize%3D8)%0A%20%20%20%20axes%5B1%5D.plot(alphas%2C%20norms%2C%20marker%3D%22o%22%2C%20label%3D%22coefficient%20norm%22)%0A%20%20%20%20axes%5B1%5D.plot(alphas%2C%20errors%2C%20marker%3D%22o%22%2C%20label%3D%22new-data%20MSE%22)%0A%20%20%20%20axes%5B1%5D.set_xscale(%22symlog%22%2C%20linthresh%3D0.01)%0A%20%20%20%20axes%5B1%5D.set(title%3D%22Too%20little%20and%20too%20much%20both%20cost%22%2C%20xlabel%3D%22L2%20strength%22)%0A%20%20%20%20axes%5B1%5D.legend()%0A%20%20%20%20plt.tight_layout()%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20More%20regularization%20is%20not%20always%20safer%0A%0A%20%20%20%20Assuming%20stronger%20regularization%20is%20automatically%20safer.%20Excessive%20constraint%20erases%20real%20signal%20and%20creates%20underfitting.%0A%0A%20%20%20%20%23%23%20Compare%20training%20and%20validation%20behavior%0A%0A%20%20%20%20Track%20training%20and%20validation%20performance%20as%20regularization%20changes.%20Inspect%20coefficient%20stability%2C%20tree%20depth%2C%20learning%20curves%2C%20or%20the%20relevant%20measure%20of%20effective%20complexity.%0A%0A%20%20%20%20%23%23%20How%20to%20choose%20its%20strength%0A%0A%20%20%20%20Choose%20regularization%20strength%20on%20deployment-like%20validation%20data%3B%20increase%20it%20for%20demonstrated%20instability%2C%20not%20as%20a%20ritual.%0A%0A%20%20%20%20%23%23%20Previous%20concept%0A%0A%20%20%20%20%5BBias%20and%20Variance%5D(%2Fconcepts%2Fbias_variance)%20explains%20the%20instability%20that%20regularization%20is%20often%20used%20to%20reduce.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
0f4be9955bb5c84337c1c47e26b6453d