import%20marimo%0A%0A__generated_with%20%3D%20%220.25.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%0A%20%20%20%20return%20(mo%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20METADATA%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22id%22%3A%20%22train_validation_test%22%2C%0A%20%20%20%20%20%20%20%20%22name%22%3A%20%22Train%2C%20Validation%2C%20and%20Test%20Sets%22%2C%0A%20%20%20%20%20%20%20%20%22kind%22%3A%20%22concept%22%2C%0A%20%20%20%20%20%20%20%20%22difficulty%22%3A%20%22beginner%22%2C%0A%20%20%20%20%20%20%20%20%22status%22%3A%20%22complete%22%2C%0A%20%20%20%20%20%20%20%20%22topics%22%3A%20%5B%22data_splitting%22%2C%20%22evaluation%22%5D%2C%0A%20%20%20%20%20%20%20%20%22related_models%22%3A%20%5B%5D%2C%0A%20%20%20%20%20%20%20%20%22prerequisites%22%3A%20%5B%5D%2C%0A%20%20%20%20%7D%0A%20%20%20%20mo.md(f%22%23%20%7BMETADATA%5B'name'%5D%7D%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20The%20rule%0A%0A%20%20%20%20Training%20data%20fits%20the%20model%2C%20validation%20data%20chooses%20the%20design%2C%20and%20test%20data%20estimates%20the%20performance%20of%20the%20finished%20design.%0A%0A%20%20%20%20%23%23%20Three%20sealed%20boxes%0A%0A%20%20%20%20Think%20of%20training%20as%20practice%2C%20validation%20as%20rehearsal%2C%20and%20testing%20as%20opening%20night.%20Repeating%20opening%20night%20until%20the%20result%20looks%20good%20turns%20the%20test%20set%20into%20another%20rehearsal.%0A%0A%20%20%20%20%23%23%20Why%20the%20roles%20must%20stay%20separate%0A%0A%20%20%20%20Model%20fitting%20and%20model%20selection%20are%20both%20forms%20of%20learning.%20If%20the%20same%20examples%20guide%20both%20choices%20and%20final%20reporting%2C%20the%20reported%20score%20becomes%20optimistic.%0A%0A%20%20%20%20%23%23%20What%20we%20will%20simulate%0A%0A%20%20%20%20**Question%3A**%20can%20validation%20choose%20model%20complexity%20without%20looking%20at%20the%20test%20set%3F%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Three%20sets%2C%20three%20different%20jobs%0A%0A%20%20%20%20Assume%20a%20dataset%20contains%20pairs%20%24(x_i%2Cy_i)%24%2C%20where%20%24x_i%24%20is%20an%20input%20and%20%24y_i%24%20is%20the%20answer%20we%20want%20to%20predict.%0A%0A%20%20%20%20%7C%20Set%20%7C%20Used%20for%20%7C%20Must%20not%20be%20used%20for%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C%0A%20%20%20%20%7C%20Training%20%7C%20Fitting%20parameters%20such%20as%20weights%20%7C%20Final%20performance%20reporting%20%7C%0A%20%20%20%20%7C%20Validation%20%7C%20Choosing%20features%2C%20thresholds%2C%20settings%2C%20and%20model%20complexity%20%7C%20Fitting%20the%20final%20reported%20score%20%7C%0A%20%20%20%20%7C%20Test%20%7C%20Estimating%20the%20performance%20of%20the%20frozen%20procedure%20%7C%20Making%20any%20further%20choice%20%7C%0A%0A%20%20%20%20The%20important%20distinction%20is%20between%20two%20kinds%20of%20learning%3A%0A%0A%20%20%20%201.%20**Parameter%20learning%3A**%20training%20data%20changes%20fitted%20parameters.%0A%20%20%20%202.%20**Design%20learning%3A**%20validation%20data%20changes%20choices%20around%20the%20fitting%20procedure.%0A%0A%20%20%20%20Both%20adapt%20the%20final%20system%20to%20data.%20This%20is%20why%20the%20test%20set%20must%20remain%20outside%20both%20loops.%0A%0A%20%20%20%20%23%23%20The%20correct%20sequence%0A%0A%20%20%20%20Suppose%20we%20compare%20polynomial%20degrees%20%241%24%2C%20%242%24%2C%20and%20%243%24.%0A%0A%20%20%20%201.%20Fit%20all%20three%20candidates%20on%20the%20training%20set.%0A%20%20%20%202.%20Measure%20all%20three%20on%20the%20validation%20set.%0A%20%20%20%203.%20Select%20the%20degree%20with%20the%20best%20validation%20result.%0A%20%20%20%204.%20Freeze%20every%20choice%2C%20including%20preprocessing%20and%20decision%20thresholds.%0A%20%20%20%205.%20Evaluate%20the%20selected%20procedure%20once%20on%20the%20test%20set.%0A%0A%20%20%20%20The%20test%20result%20estimates%20the%20performance%20of%20the%20**whole%20selection%20procedure**%2C%20not%20only%20one%20fitted%20set%20of%20weights.%0A%0A%20%20%20%20%23%23%20A%20concrete%20split%0A%0A%20%20%20%20With%20%241%7B%2C%7D000%24%20examples%2C%20a%20simple%20split%20could%20be%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20600%5Ctext%7B%20train%7D%2C%5Cqquad200%5Ctext%7B%20validation%7D%2C%5Cqquad200%5Ctext%7B%20test%7D.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20These%20numbers%20are%20not%20universal.%20The%20split%20must%20preserve%20the%20structure%20of%20the%20real%20prediction%20problem%3A%0A%0A%20%20%20%20-%20use%20a%20**time%20split**%20when%20predicting%20the%20future%3B%0A%20%20%20%20-%20use%20a%20**group%20split**%20when%20the%20same%20person%2C%20device%2C%20or%20document%20can%20appear%20many%20times%3B%0A%20%20%20%20-%20use%20**stratification**%20when%20a%20rare%20label%20should%20remain%20represented%20in%20every%20set.%0A%0A%20%20%20%20The%20rule%20is%20not%20%E2%80%9Calways%20use%2060%2F20%2F20.%E2%80%9D%20The%20rule%20is%20%E2%80%9Cmake%20evaluation%20resemble%20future%20use.%E2%80%9D%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%20%20%20%20from%20sklearn.linear_model%20import%20LinearRegression%0A%20%20%20%20from%20sklearn.metrics%20import%20mean_squared_error%0A%20%20%20%20from%20sklearn.pipeline%20import%20make_pipeline%0A%20%20%20%20from%20sklearn.preprocessing%20import%20PolynomialFeatures%0A%0A%20%20%20%20rng%20%3D%20np.random.default_rng(12)%0A%20%20%20%20x%20%3D%20rng.uniform(-3%2C%203%2C%20180)%0A%20%20%20%20y%20%3D%20np.sin(x)%20%2B%200.18%20*%20x%20%2B%20rng.normal(0%2C%200.28%2C%20len(x))%0A%20%20%20%20order%20%3D%20rng.permutation(len(x))%0A%20%20%20%20train_idx%2C%20validation_idx%2C%20test_idx%20%3D%20order%5B%3A90%5D%2C%20order%5B90%3A135%5D%2C%20order%5B135%3A%5D%0A%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20LinearRegression%2C%0A%20%20%20%20%20%20%20%20PolynomialFeatures%2C%0A%20%20%20%20%20%20%20%20make_pipeline%2C%0A%20%20%20%20%20%20%20%20mean_squared_error%2C%0A%20%20%20%20%20%20%20%20plt%2C%0A%20%20%20%20%20%20%20%20test_idx%2C%0A%20%20%20%20%20%20%20%20train_idx%2C%0A%20%20%20%20%20%20%20%20validation_idx%2C%0A%20%20%20%20%20%20%20%20x%2C%0A%20%20%20%20%20%20%20%20y%2C%0A%20%20%20%20)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20indices%20above%20create%20three%20disjoint%20groups%20before%20any%20candidate%20is%20fitted.%20The%0A%20%20%20%20next%20cell%20uses%20training%20data%20to%20fit%2C%20validation%20data%20to%20choose%20the%20polynomial%0A%20%20%20%20degree%2C%20and%20test%20data%20exactly%20once%20after%20that%20choice%20has%20been%20made.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20LinearRegression%2C%0A%20%20%20%20PolynomialFeatures%2C%0A%20%20%20%20make_pipeline%2C%0A%20%20%20%20mean_squared_error%2C%0A%20%20%20%20plt%2C%0A%20%20%20%20test_idx%2C%0A%20%20%20%20train_idx%2C%0A%20%20%20%20validation_idx%2C%0A%20%20%20%20x%2C%0A%20%20%20%20y%2C%0A)%3A%0A%20%20%20%20degrees%20%3D%20list(range(1%2C%2016))%0A%20%20%20%20train_error%2C%20validation_error%2C%20candidates%20%3D%20%5B%5D%2C%20%5B%5D%2C%20%5B%5D%0A%20%20%20%20for%20degree%20in%20degrees%3A%0A%20%20%20%20%20%20%20%20candidate%20%3D%20make_pipeline(PolynomialFeatures(degree)%2C%20LinearRegression())%0A%20%20%20%20%20%20%20%20candidate.fit(x%5Btrain_idx%2C%20None%5D%2C%20y%5Btrain_idx%5D)%0A%20%20%20%20%20%20%20%20candidates.append(candidate)%0A%20%20%20%20%20%20%20%20train_error.append(mean_squared_error(y%5Btrain_idx%5D%2C%20candidate.predict(x%5Btrain_idx%2C%20None%5D)))%0A%20%20%20%20%20%20%20%20validation_error.append(mean_squared_error(y%5Bvalidation_idx%5D%2C%20candidate.predict(x%5Bvalidation_idx%2C%20None%5D)))%0A%0A%20%20%20%20best_index%20%3D%20int(min(range(len(degrees))%2C%20key%3Dvalidation_error.__getitem__))%0A%20%20%20%20best_degree%20%3D%20degrees%5Bbest_index%5D%0A%20%20%20%20final_model%20%3D%20candidates%5Bbest_index%5D%0A%20%20%20%20test_error%20%3D%20mean_squared_error(y%5Btest_idx%5D%2C%20final_model.predict(x%5Btest_idx%2C%20None%5D))%0A%0A%20%20%20%20fig%2C%20axes%20%3D%20plt.subplots(1%2C%202%2C%20figsize%3D(11%2C%204))%0A%20%20%20%20axes%5B0%5D.scatter(x%5Btrain_idx%5D%2C%20y%5Btrain_idx%5D%2C%20label%3D%22train%22%2C%20alpha%3D0.7)%0A%20%20%20%20axes%5B0%5D.scatter(x%5Bvalidation_idx%5D%2C%20y%5Bvalidation_idx%5D%2C%20label%3D%22validation%22%2C%20alpha%3D0.7)%0A%20%20%20%20axes%5B0%5D.scatter(x%5Btest_idx%5D%2C%20y%5Btest_idx%5D%2C%20label%3D%22test%20%E2%80%94%20untouched%22%2C%20alpha%3D0.7)%0A%20%20%20%20axes%5B0%5D.set(title%3D%22Three%20datasets%2C%20three%20jobs%22%2C%20xlabel%3D%22x%22%2C%20ylabel%3D%22target%22)%0A%20%20%20%20axes%5B0%5D.legend()%0A%20%20%20%20axes%5B1%5D.plot(degrees%2C%20train_error%2C%20marker%3D%22o%22%2C%20label%3D%22train%20error%22)%0A%20%20%20%20axes%5B1%5D.plot(degrees%2C%20validation_error%2C%20marker%3D%22o%22%2C%20label%3D%22validation%20error%22)%0A%20%20%20%20axes%5B1%5D.axvline(best_degree%2C%20color%3D%22crimson%22%2C%20linestyle%3D%22--%22%2C%20label%3Df%22chosen%20degree%20%3D%20%7Bbest_degree%7D%22)%0A%20%20%20%20axes%5B1%5D.set(title%3Df%22Test%20MSE%20reported%20once%3A%20%7Btest_error%3A.3f%7D%22%2C%20xlabel%3D%22polynomial%20degree%22%2C%20ylabel%3D%22MSE%22)%0A%20%20%20%20axes%5B1%5D.legend()%0A%20%20%20%20plt.tight_layout()%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20How%20the%20test%20set%20quietly%20becomes%20training%20data%0A%0A%20%20%20%20Trying%20many%20alternatives%20against%20the%20test%20set%20and%20reporting%20only%20the%20winner%20silently%20trains%20the%20project%20on%20its%20test%20data.%0A%0A%20%20%20%20%23%23%20Trace%20every%20decision%20back%20to%20its%20data%0A%0A%20%20%20%20Write%20down%20which%20decisions%20each%20dataset%20influenced.%20Check%20whether%20preprocessing%2C%20feature%20selection%2C%20thresholds%2C%20and%20hyperparameters%20were%20fixed%20before%20the%20test%20set%20was%20opened.%0A%0A%20%20%20%20%23%23%20The%20clean%20sequence%0A%0A%20%20%20%20Design%20the%20split%20to%20imitate%20deployment%2C%20choose%20with%20validation%20data%2C%20and%20use%20the%20test%20set%20once%20the%20entire%20modeling%20procedure%20is%20frozen.%0A%0A%20%20%20%20%23%23%20Next%20concept%0A%0A%20%20%20%20%5BGeneralization%5D(%2Fconcepts%2Fgeneralization)%20explains%20what%20the%20untouched%20evaluation%20set%20is%20trying%20to%20measure.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
d7239696e5502186981d5dd9a2735c9e