import%20marimo%0A%0A__generated_with%20%3D%20%220.25.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%0A%20%20%20%20return%20(mo%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20METADATA%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22id%22%3A%20%22metrics%22%2C%0A%20%20%20%20%20%20%20%20%22name%22%3A%20%22Metrics%22%2C%0A%20%20%20%20%20%20%20%20%22kind%22%3A%20%22concept%22%2C%0A%20%20%20%20%20%20%20%20%22difficulty%22%3A%20%22beginner%22%2C%0A%20%20%20%20%20%20%20%20%22status%22%3A%20%22complete%22%2C%0A%20%20%20%20%20%20%20%20%22topics%22%3A%20%5B%22classification_metrics%22%2C%20%22evaluation%22%5D%2C%0A%20%20%20%20%20%20%20%20%22related_models%22%3A%20%5B%5D%2C%0A%20%20%20%20%20%20%20%20%22prerequisites%22%3A%20%5B%22train_validation_test%22%5D%2C%0A%20%20%20%20%7D%0A%20%20%20%20mo.md(f%22%23%20%7BMETADATA%5B'name'%5D%7D%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20The%20central%20idea%0A%0A%20%20%20%20A%20metric%20compresses%20model%20behavior%20into%20a%20number%2C%20so%20choosing%20one%20means%20choosing%20which%20mistakes%20matter.%0A%0A%20%20%20%20%23%23%20Different%20mistakes%20have%20different%20costs%0A%0A%20%20%20%20A%20metric%20is%20a%20lens%2C%20not%20a%20verdict.%20Precision%20looks%20at%20predicted%20positives%2C%20recall%20looks%20at%20real%20positives%2C%20and%20neither%20says%20what%20operating%20threshold%20the%20product%20can%20afford.%0A%0A%20%20%20%20%23%23%20Why%20no%20metric%20is%20universal%0A%0A%20%20%20%20Two%20models%20can%20have%20the%20same%20accuracy%20while%20creating%20radically%20different%20consequences%20when%20positives%20are%20rare%20or%20one%20type%20of%20error%20is%20expensive.%0A%0A%20%20%20%20%23%23%20What%20we%20will%20measure%0A%0A%20%20%20%20**Question%3A**%20how%20does%20moving%20one%20classification%20threshold%20change%20the%20apparent%20quality%20of%20the%20same%20underlying%20scores%3F%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Begin%20with%20four%20counts%0A%0A%20%20%20%20For%20binary%20decisions%2C%20every%20example%20belongs%20to%20one%20cell%20of%20a%20confusion%20matrix%3A%0A%0A%20%20%20%20%7C%20%7C%20Predicted%20positive%20%7C%20Predicted%20negative%20%7C%0A%20%20%20%20%7C---%7C---%3A%7C---%3A%7C%0A%20%20%20%20%7C%20Actually%20positive%20%7C%20true%20positive%20(%24TP%24)%20%7C%20false%20negative%20(%24FN%24)%20%7C%0A%20%20%20%20%7C%20Actually%20negative%20%7C%20false%20positive%20(%24FP%24)%20%7C%20true%20negative%20(%24TN%24)%20%7C%0A%0A%20%20%20%20All%20common%20classification%20metrics%20select%20and%20combine%20these%20four%20counts.%0A%0A%20%20%20%20%23%23%20Main%20formulas%0A%0A%20%20%20%20%23%23%23%20Accuracy%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Ctext%7Baccuracy%7D%3D%5Cfrac%7BTP%2BTN%7D%7BTP%2BTN%2BFP%2BFN%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Accuracy%20asks%3A%20%E2%80%9Cwhat%20fraction%20of%20all%20decisions%20were%20correct%3F%E2%80%9D%20It%20can%20hide%20failure%20on%20a%20rare%20class.%0A%0A%20%20%20%20%23%23%23%20Precision%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Ctext%7Bprecision%7D%3D%5Cfrac%7BTP%7D%7BTP%2BFP%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Precision%20asks%3A%20%E2%80%9Camong%20predicted%20positives%2C%20how%20many%20were%20truly%20positive%3F%E2%80%9D%20It%20matters%20when%20false%20alarms%20are%20costly.%0A%0A%20%20%20%20%23%23%23%20Recall%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Ctext%7Brecall%7D%3D%5Cfrac%7BTP%7D%7BTP%2BFN%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Recall%20asks%3A%20%E2%80%9Camong%20real%20positives%2C%20how%20many%20were%20found%3F%E2%80%9D%20It%20matters%20when%20missed%20positives%20are%20costly.%0A%0A%20%20%20%20%23%23%23%20F1%20score%0A%0A%20%20%20%20%24%24%0A%20%20%20%20F_1%3D2%5Cfrac%7B%5Ctext%7Bprecision%7D%5Ctimes%5Ctext%7Brecall%7D%7D%0A%20%20%20%20%7B%5Ctext%7Bprecision%7D%2B%5Ctext%7Brecall%7D%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20F1%20is%20the%20harmonic%20mean.%20It%20is%20high%20only%20when%20both%20precision%20and%20recall%20are%20high%2C%20but%20it%20does%20not%20include%20true%20negatives.%0A%0A%20%20%20%20%23%23%20A%20numerical%20example%0A%0A%20%20%20%20Suppose%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20TP%3D30%2C%5Cquad%20FP%3D10%2C%5Cquad%20FN%3D20%2C%5Cquad%20TN%3D140.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Then%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Ctext%7Baccuracy%7D%3D%5Cfrac%7B30%2B140%7D%7B200%7D%3D0.85%2C%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Ctext%7Bprecision%7D%3D%5Cfrac%7B30%7D%7B30%2B10%7D%3D0.75%2C%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Ctext%7Brecall%7D%3D%5Cfrac%7B30%7D%7B30%2B20%7D%3D0.60.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20same%20decisions%20can%20therefore%20look%20strong%20by%20accuracy%20and%20weaker%20by%20recall.%0A%0A%20%20%20%20%23%23%20A%20score%20is%20not%20yet%20a%20decision%0A%0A%20%20%20%20A%20binary%20score%20becomes%20a%20class%20only%20after%20choosing%20a%20threshold%20%24t%24%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Chat%20y%3D%5Cmathbb%201%5Bs%5Cgeq%20t%5D.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Lowering%20%24t%24%20usually%20predicts%20more%20positives%3A%20recall%20rises%20and%20precision%20may%20fall.%20Raising%20%24t%24%20usually%20does%20the%20opposite.%20This%20is%20a%20change%20in%20operating%20policy%2C%20not%20a%20new%20set%20of%20underlying%20scores.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%20%20%20%20from%20sklearn.metrics%20import%20confusion_matrix%2C%20precision_recall_fscore_support%0A%0A%20%20%20%20rng%20%3D%20np.random.default_rng(9)%0A%20%20%20%20y_true%20%3D%20np.r_%5Bnp.zeros(160%2C%20dtype%3Dint)%2C%20np.ones(40%2C%20dtype%3Dint)%5D%0A%20%20%20%20scores%20%3D%20np.r_%5Brng.beta(2%2C%206%2C%20160)%2C%20rng.beta(5%2C%202.8%2C%2040)%5D%0A%20%20%20%20thresholds%20%3D%20np.linspace(0.05%2C%200.95%2C%2080)%0A%20%20%20%20precision%2C%20recall%2C%20f1%20%3D%20%5B%5D%2C%20%5B%5D%2C%20%5B%5D%0A%20%20%20%20for%20threshold%20in%20thresholds%3A%0A%20%20%20%20%20%20%20%20predicted%20%3D%20scores%20%3E%3D%20threshold%0A%20%20%20%20%20%20%20%20p%2C%20r%2C%20f%2C%20_%20%3D%20precision_recall_fscore_support(y_true%2C%20predicted%2C%20average%3D%22binary%22%2C%20zero_division%3D0)%0A%20%20%20%20%20%20%20%20precision.append(p)%0A%20%20%20%20%20%20%20%20recall.append(r)%0A%20%20%20%20%20%20%20%20f1.append(f)%0A%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20confusion_matrix%2C%0A%20%20%20%20%20%20%20%20f1%2C%0A%20%20%20%20%20%20%20%20np%2C%0A%20%20%20%20%20%20%20%20plt%2C%0A%20%20%20%20%20%20%20%20precision%2C%0A%20%20%20%20%20%20%20%20recall%2C%0A%20%20%20%20%20%20%20%20scores%2C%0A%20%20%20%20%20%20%20%20thresholds%2C%0A%20%20%20%20%20%20%20%20y_true%2C%0A%20%20%20%20)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20scores%20stay%20fixed%20in%20the%20next%20experiment.%20Only%20the%20decision%20threshold%20moves.%0A%20%20%20%20This%20isolates%20the%20important%20fact%3A%20precision%2C%20recall%2C%20and%20F1%20can%20change%20even%20when%0A%20%20%20%20the%20underlying%20scoring%20rule%20has%20not%20been%20retrained.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20confusion_matrix%2C%0A%20%20%20%20f1%2C%0A%20%20%20%20np%2C%0A%20%20%20%20plt%2C%0A%20%20%20%20precision%2C%0A%20%20%20%20recall%2C%0A%20%20%20%20scores%2C%0A%20%20%20%20thresholds%2C%0A%20%20%20%20y_true%2C%0A)%3A%0A%20%20%20%20chosen%20%3D%200.5%0A%20%20%20%20matrix%20%3D%20confusion_matrix(y_true%2C%20scores%20%3E%3D%20chosen)%0A%20%20%20%20fig%2C%20axes%20%3D%20plt.subplots(1%2C%202%2C%20figsize%3D(10%2C%204))%0A%20%20%20%20axes%5B0%5D.plot(thresholds%2C%20precision%2C%20label%3D%22precision%22)%0A%20%20%20%20axes%5B0%5D.plot(thresholds%2C%20recall%2C%20label%3D%22recall%22)%0A%20%20%20%20axes%5B0%5D.plot(thresholds%2C%20f1%2C%20label%3D%22F1%22)%0A%20%20%20%20axes%5B0%5D.axvline(chosen%2C%20color%3D%22black%22%2C%20linestyle%3D%22--%22%2C%20label%3D%22chosen%20threshold%22)%0A%20%20%20%20axes%5B0%5D.set(title%3D%22One%20score%20model%2C%20many%20operating%20points%22%2C%20xlabel%3D%22threshold%22%2C%20ylabel%3D%22metric%22)%0A%20%20%20%20axes%5B0%5D.legend()%0A%20%20%20%20image%20%3D%20axes%5B1%5D.imshow(matrix%2C%20cmap%3D%22Blues%22)%0A%20%20%20%20for%20row%2C%20column%20in%20np.ndindex(matrix.shape)%3A%0A%20%20%20%20%20%20%20%20axes%5B1%5D.text(column%2C%20row%2C%20matrix%5Brow%2C%20column%5D%2C%20ha%3D%22center%22%2C%20va%3D%22center%22)%0A%20%20%20%20axes%5B1%5D.set(title%3D%22Confusion%20matrix%20at%20threshold%200.50%22%2C%20xlabel%3D%22predicted%22%2C%20ylabel%3D%22actual%22)%0A%20%20%20%20axes%5B1%5D.set_xticks(%5B0%2C%201%5D%2C%20%5B%22negative%22%2C%20%22positive%22%5D)%0A%20%20%20%20axes%5B1%5D.set_yticks(%5B0%2C%201%5D%2C%20%5B%22negative%22%2C%20%22positive%22%5D)%0A%20%20%20%20fig.colorbar(image%2C%20ax%3Daxes%5B1%5D%2C%20fraction%3D0.046)%0A%20%20%20%20plt.tight_layout()%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20The%20accuracy%20trap%0A%0A%20%20%20%20Selecting%20accuracy%20by%20habit%2C%20especially%20when%20the%20positive%20class%20is%20rare%2C%20or%20choosing%20a%20threshold%20on%20the%20final%20test%20set.%0A%0A%20%20%20%20%23%23%20Return%20to%20the%20confusion%20counts%0A%0A%20%20%20%20Inspect%20the%20confusion%20matrix%2C%20precision%E2%80%93recall%20behavior%2C%20calibration%2C%20and%20performance%20across%20important%20groups.%20Translate%20false%20positives%20and%20false%20negatives%20into%20real%20costs.%0A%0A%20%20%20%20%23%23%20Choose%20from%20the%20real%20decision%0A%0A%20%20%20%20Choose%20the%20metric%20from%20the%20decision%20cost%2C%20then%20report%20complementary%20diagnostics%20and%20select%20the%20operating%20threshold%20on%20validation%20data.%0A%0A%20%20%20%20%23%23%20Previous%20and%20next%0A%0A%20%20%20%20-%20Previous%3A%20%5BData%20Leakage%5D(%2Fconcepts%2Fdata_leakage)%20ensures%20that%20the%20evaluation%20evidence%20is%20valid.%0A%20%20%20%20-%20Next%3A%20%5BThe%20Perceptron%5D(%2Fconcepts%2Fperceptron)%20shows%20how%20a%20binary%20decision%20score%20can%20be%20learned.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
ad0663283c254950085d73f6a2cbc6eb