import%20marimo%0A%0A__generated_with%20%3D%20%220.25.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%0A%20%20%20%20return%20(mo%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20METADATA%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22id%22%3A%20%22training_loop%22%2C%0A%20%20%20%20%20%20%20%20%22name%22%3A%20%22Batches%2C%20Epochs%2C%20and%20Early%20Stopping%22%2C%0A%20%20%20%20%20%20%20%20%22kind%22%3A%20%22concept%22%2C%0A%20%20%20%20%20%20%20%20%22difficulty%22%3A%20%22beginner%22%2C%0A%20%20%20%20%20%20%20%20%22status%22%3A%20%22complete%22%2C%0A%20%20%20%20%20%20%20%20%22summary%22%3A%20%22An%20epoch%20is%20one%20complete%20pass%20through%20the%20training%20data%3B%20batches%20split%20that%20pass%20into%20successive%20weight%20updates.%22%2C%0A%20%20%20%20%20%20%20%20%22topics%22%3A%20%5B%22neural_networks%22%2C%20%22optimization%22%2C%20%22evaluation%22%5D%2C%0A%20%20%20%20%20%20%20%20%22related_models%22%3A%20%5B%22mlp%22%2C%20%22cnn%22%5D%2C%0A%20%20%20%20%20%20%20%20%22prerequisites%22%3A%20%5B%22train_validation_test%22%2C%20%22loss_optimization%22%5D%2C%0A%20%20%20%20%7D%0A%20%20%20%20mo.md(f%22%23%20%7BMETADATA%5B'name'%5D%7D%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20In%20one%20sentence%0A%0A%20%20%20%20An%20**epoch**%20is%20one%20complete%20pass%20through%20the%20training%20data%3B%20**batches**%20split%20that%20pass%20into%20smaller%20weight%20updates%2C%20while%20**early%20stopping**%20stops%20training%20when%20validation%20performance%20no%20longer%20improves.%0A%0A%20%20%20%20%23%23%20Four%20terms%20to%20distinguish%0A%0A%20%20%20%20%7C%20Term%20%7C%20What%20the%20model%20sees%20%7C%20What%20happens%20%7C%0A%20%20%20%20%7C---%7C---%7C---%7C%0A%20%20%20%20%7C%20Example%20%7C%20one%20image%20or%20data%20row%20%7C%20one%20prediction%20contributes%20to%20the%20loss%20%7C%0A%20%20%20%20%7C%20Batch%20%7C%20a%20small%20group%20of%20examples%20%7C%20the%20weights%20are%20updated%20after%20computing%20the%20batch%20loss%20%7C%0A%20%20%20%20%7C%20Epoch%20%7C%20every%20training%20example%2C%20once%20%7C%20several%20updates%2C%20one%20per%20batch%20%7C%0A%20%20%20%20%7C%20Iteration%20%2F%20step%20%7C%20one%20batch%20%7C%20usually%20one%20optimizer%20update%20%7C%0A%0A%20%20%20%20With%2010%2C000%20images%20and%20a%20batch%20size%20of%20100%2C%20one%20epoch%20contains%20%2410%5C%2C000%20%2F%20100%20%3D%20100%24%20steps.%20After%2010%20epochs%2C%20each%20image%20has%20been%20used%20about%2010%20times%20(the%20order%20is%20usually%20shuffled%20before%20each%20epoch).%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_()%3A%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20from%20matplotlib.patches%20import%20FancyBboxPatch%0A%0A%20%20%20%20return%20FancyBboxPatch%2C%20np%2C%20plt%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(FancyBboxPatch%2C%20np%2C%20plt)%3A%0A%20%20%20%20paper%2C%20ink%2C%20blue%2C%20coral%2C%20gold%20%3D%20%22%23FAF7F0%22%2C%20%22%2317324D%22%2C%20%22%233A7CA5%22%2C%20%22%23E76F51%22%2C%20%22%23E9C46A%22%0A%20%20%20%20fig%2C%20axes%20%3D%20plt.subplots(1%2C%202%2C%20figsize%3D(14%2C%204.7)%2C%20facecolor%3Dpaper)%0A%20%20%20%20for%20ax%20in%20axes%3A%0A%20%20%20%20%20%20%20%20ax.set_facecolor(paper)%0A%0A%20%20%20%20ax%20%3D%20axes%5B0%5D%0A%20%20%20%20ax.set(xlim%3D(0%2C%2011.2)%2C%20ylim%3D(0%2C%205.6))%0A%20%20%20%20ax.axis(%22off%22)%0A%20%20%20%20batches%20%3D%20%5B(0.7%2C%202.6%2C%20blue%2C%20%22batch%201%5Cn100%20images%22)%2C%20(3.3%2C%202.6%2C%20gold%2C%20%22batch%202%5Cn100%20images%22)%2C%20(5.9%2C%202.6%2C%20blue%2C%20%22batch%203%5Cn100%20images%22)%2C%20(8.5%2C%202.6%2C%20gold%2C%20%22%E2%80%A6%22)%5D%0A%20%20%20%20for%20x%2C%20y%2C%20color%2C%20label%20in%20batches%3A%0A%20%20%20%20%20%20%20%20ax.add_patch(FancyBboxPatch((x%2C%20y)%2C%201.8%2C%201.15%2C%20boxstyle%3D%22round%2Cpad%3D0.13%22%2C%20facecolor%3Dcolor%2C%20edgecolor%3Dink%2C%20linewidth%3D1.1))%0A%20%20%20%20%20%20%20%20ax.text(x%20%2B%200.9%2C%20y%20%2B%200.58%2C%20label%2C%20ha%3D%22center%22%2C%20va%3D%22center%22%2C%20color%3D%22white%22%20if%20color%20%3D%3D%20blue%20else%20ink%2C%20fontsize%3D10%2C%20weight%3D%22bold%22)%0A%20%20%20%20for%20x%20in%20%5B2.55%2C%205.15%2C%207.75%5D%3A%0A%20%20%20%20%20%20%20%20ax.annotate(%22%22%2C%20xy%3D(x%20%2B%200.55%2C%203.18)%2C%20xytext%3D(x%2C%203.18)%2C%20arrowprops%3D%7B%22arrowstyle%22%3A%20%22-%3E%22%2C%20%22color%22%3A%20ink%2C%20%22lw%22%3A%201.5%7D)%0A%20%20%20%20ax.text(5.5%2C%204.65%2C%20%221%20epoch%20%3D%20every%20batch%20processed%20once%22%2C%20ha%3D%22center%22%2C%20color%3Dink%2C%20fontsize%3D13%2C%20weight%3D%22bold%22)%0A%20%20%20%20ax.text(5.5%2C%201.55%2C%20%22each%20batch%3A%20prediction%20%E2%86%92%20loss%20%E2%86%92%20gradients%20%E2%86%92%20update%22%2C%20ha%3D%22center%22%2C%20color%3Dcoral%2C%20fontsize%3D10)%0A%20%20%20%20ax.set_title(%22A%20full%20pass%20is%20split%20into%20batches%22%2C%20loc%3D%22left%22%2C%20color%3Dink%2C%20fontsize%3D14%2C%20pad%3D12)%0A%0A%20%20%20%20ax%20%3D%20axes%5B1%5D%0A%20%20%20%20epochs%20%3D%20np.arange(1%2C%2031)%0A%20%20%20%20train_loss%20%3D%201.1%20*%20np.exp(-epochs%20%2F%2010)%20%2B%200.08%0A%20%20%20%20validation_loss%20%3D%201.05%20*%20np.exp(-epochs%20%2F%209)%20%2B%200.12%20%2B%200.0035%20*%20np.maximum(epochs%20-%2016%2C%200)%20**%201.45%0A%20%20%20%20best%20%3D%20int(np.argmin(validation_loss))%0A%20%20%20%20ax.plot(epochs%2C%20train_loss%2C%20color%3Dblue%2C%20linewidth%3D2.5%2C%20label%3D%22training%22)%0A%20%20%20%20ax.plot(epochs%2C%20validation_loss%2C%20color%3Dcoral%2C%20linewidth%3D2.5%2C%20label%3D%22validation%22)%0A%20%20%20%20ax.axvline(epochs%5Bbest%5D%2C%20color%3Dgold%2C%20linestyle%3D%22--%22%2C%20linewidth%3D2)%0A%20%20%20%20ax.scatter(%5Bepochs%5Bbest%5D%5D%2C%20%5Bvalidation_loss%5Bbest%5D%5D%2C%20color%3Dgold%2C%20zorder%3D3%2C%20s%3D55)%0A%20%20%20%20ax.annotate(f%22best%20validation%5Cnepoch%20%7Bepochs%5Bbest%5D%7D%22%2C%20xy%3D(epochs%5Bbest%5D%2C%20validation_loss%5Bbest%5D)%2C%20xytext%3D(epochs%5Bbest%5D%20%2B%202%2C%200.78)%2C%20arrowprops%3D%7B%22arrowstyle%22%3A%20%22-%3E%22%2C%20%22color%22%3A%20ink%7D%2C%20color%3Dink%2C%20fontsize%3D10)%0A%20%20%20%20ax.set(xlabel%3D%22epoch%22%2C%20ylabel%3D%22loss%22%2C%20xlim%3D(1%2C%2030))%0A%20%20%20%20ax.grid(color%3D%22%23DCE3E8%22)%0A%20%20%20%20ax.set_axisbelow(True)%0A%20%20%20%20ax.spines%5B%5B%22top%22%2C%20%22right%22%5D%5D.set_visible(False)%0A%20%20%20%20ax.legend(frameon%3DFalse)%0A%20%20%20%20ax.set_title(%22Early%20stopping%3A%20keep%20the%20best%20model%22%2C%20loc%3D%22left%22%2C%20color%3Dink%2C%20fontsize%3D14%2C%20pad%3D12)%0A%0A%20%20%20%20fig.suptitle(%22Training%20cadence%3A%20batches%2C%20epochs%2C%20then%20stopping%22%2C%20x%3D0.055%2C%20y%3D1.04%2C%20ha%3D%22left%22%2C%20color%3Dink%2C%20fontsize%3D18%2C%20weight%3D%22bold%22)%0A%20%20%20%20plt.tight_layout()%0A%20%20%20%20fig%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Why%20use%20batches%3F%0A%0A%20%20%20%20Computing%20a%20gradient%20over%20the%20entire%20dataset%20before%20every%20update%20is%20precise%2C%20but%20slow%20and%20often%20too%20memory%20intensive.%20A%20batch%20estimates%20the%20gradient%20from%20much%20less%20data%3A%20the%20estimate%20is%20noisier%2C%20but%20it%20allows%20more%20frequent%20updates%20and%20runs%20efficiently%20on%20GPUs.%0A%0A%20%20%20%20Batch%20size%20involves%20a%20tradeoff%3A%0A%0A%20%20%20%20-%20**Small%20batch**%3A%20less%20memory%2C%20noisier%20gradients%2C%20and%20more%20steps%20per%20epoch.%0A%20%20%20%20-%20**Large%20batch**%3A%20better%20hardware%20throughput%20and%20more%20stable%20gradients%2C%20but%20more%20memory%20and%20fewer%20updates%20per%20epoch.%0A%0A%20%20%20%20Batch%20size%20alone%20does%20not%20determine%20quality%3A%20learning%20rate%2C%20data%20augmentation%2C%20regularization%2C%20and%20architecture%20all%20interact%20with%20it.%0A%0A%20%20%20%20%23%23%20Training%20versus%20validation%0A%0A%20%20%20%20Training%20data%20is%20used%20to%20update%20the%20weights.%20Validation%20data%20does%20not%20update%20them%3A%20it%20measures%20how%20well%20the%20model%20generalizes%20to%20unseen%20examples.%20Loss%20and%20a%20task%20specific%20metric%20are%20often%20tracked%20after%20each%20epoch.%0A%0A%20%20%20%20Keep%20the%20final%20test%20set%20separate%3A%20use%20it%20for%20the%20final%20estimate%2C%20not%20to%20decide%20when%20to%20stop%20training.%0A%0A%20%20%20%20%23%23%20How%20early%20stopping%20works%0A%0A%20%20%20%20Save%20the%20weights%20whenever%20the%20validation%20metric%20improves.%20If%20it%20does%20not%20improve%20for%20a%20set%20number%20of%20epochs%2C%20called%20*patience*%2C%20stop%20training%20and%20restore%20the%20best%20saved%20weights.%0A%0A%20%20%20%20This%20does%20not%20replace%20careful%20analysis%3A%20a%20validation%20set%20that%20is%20too%20small%20or%20poorly%20separated%20can%20give%20a%20misleading%20signal.%20See%20%5BTrain%2C%20Validation%20and%20Test%20Sets%5D(%2Fconcepts%2Ftrain_validation_test)%20for%20data%20splitting.%0A%0A%20%20%20%20%23%23%20Key%20takeaway%0A%0A%20%20%20%20Never%20compare%20experiments%20by%20epoch%20count%20alone.%20Also%20compare%20dataset%20size%2C%20batch%20size%2C%20total%20steps%2C%20learning%20rate%2C%20and%20validation%20curves.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
63e085d110457ae7785be1b6786d5b8a