import%20marimo%0A%0A__generated_with%20%3D%20%220.25.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%0A%20%20%20%20return%20(mo%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20METADATA%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22id%22%3A%20%22random_forest%22%2C%0A%20%20%20%20%20%20%20%20%22name%22%3A%20%22Random%20Forest%22%2C%0A%20%20%20%20%20%20%20%20%22types%22%3A%20%5B%22algorithm%22%5D%2C%0A%20%20%20%20%20%20%20%20%22families%22%3A%20%5B%22trees%22%2C%20%22ensembles%22%5D%2C%0A%20%20%20%20%20%20%20%20%22tasks%22%3A%20%5B%22classification%22%2C%20%22regression%22%5D%2C%0A%20%20%20%20%20%20%20%20%22data%22%3A%20%5B%22tabular%22%5D%2C%0A%20%20%20%20%20%20%20%20%22learning%22%3A%20%5B%22supervised%22%5D%2C%0A%20%20%20%20%20%20%20%20%22capacity%22%3A%20%22non_parametric%22%2C%0A%20%20%20%20%20%20%20%20%22mechanisms%22%3A%20%5B%22bagging%22%2C%20%22random_subspaces%22%2C%20%22threshold_rules%22%5D%2C%0A%20%20%20%20%20%20%20%20%22properties%22%3A%20%5B%22ensemble%22%2C%20%22nonlinear%22%5D%2C%0A%20%20%20%20%20%20%20%20%22constraints%22%3A%20%5B%22poor_extrapolation%22%5D%2C%0A%20%20%20%20%20%20%20%20%22difficulty%22%3A%20%22beginner%22%2C%0A%20%20%20%20%20%20%20%20%22status%22%3A%20%22complete%22%2C%0A%20%20%20%20%20%20%20%20%22explainability%22%3A%20%22medium%22%2C%0A%20%20%20%20%20%20%20%20%22training_cost%22%3A%20%22medium%22%2C%0A%20%20%20%20%20%20%20%20%22inference_cost%22%3A%20%22medium%22%2C%0A%20%20%20%20%20%20%20%20%22data_appetite%22%3A%20%22medium%22%2C%0A%20%20%20%20%7D%0A%0A%20%20%20%20mo.md(f%22%23%20%7BMETADATA%5B'name'%5D%7D%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20The%20ensemble%20idea%0A%0A%20%20%20%20Random%20Forest%20averages%20many%20deliberately%20varied%20decision%20trees%20to%20create%20a%20more%20stable%20predictor.%0A%0A%20%20%20%20%23%23%20Let%20many%20different%20trees%20vote%0A%0A%20%20%20%20Ask%20many%20competent%20but%20imperfect%20reviewers%20to%20judge%20slightly%20different%20evidence%2C%20then%20combine%20their%20answers.%20Averaging%20helps%20only%20when%20reviewers%20are%20not%20making%20exactly%20the%20same%20mistakes.%0A%0A%20%20%20%20Random%20Forest%20creates%20diversity%20by%20training%20each%20tree%20on%20a%20bootstrap%20sample%20and%20considering%20a%20random%20subset%20of%20features%20at%20each%20split.%0A%0A%20%20%20%20%23%23%20Input%20and%20output%0A%0A%20%20%20%20-%20**Input%3A**%20primarily%20tabular%20features.%0A%20%20%20%20-%20**Output%3A**%20averaged%20numeric%20prediction%20or%20class%20vote%2Fprobability.%0A%20%20%20%20-%20**Learns%3A**%20many%20decision%20trees.%0A%0A%20%20%20%20%23%23%20How%20it%20works%0A%0A%20%20%20%201.%20Sample%20training%20rows%20with%20replacement%20for%20each%20tree.%0A%20%20%20%202.%20Grow%20a%20tree%2C%20considering%20only%20a%20random%20feature%20subset%20at%20each%20split.%0A%20%20%20%203.%20Repeat%20independently.%0A%20%20%20%204.%20Average%20regression%20outputs%20or%20class%20probabilities.%0A%0A%20%20%20%20Deep%20individual%20trees%20have%20low%20bias%20but%20high%20variance.%20Averaging%20reduces%20variance%20while%20retaining%20nonlinear%20thresholds%20and%20interactions.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Step%20by%20step%3A%20build%20and%20combine%20the%20trees%0A%0A%20%20%20%20%23%23%23%201.%20Bootstrap%20the%20rows%0A%0A%20%20%20%20For%20a%20training%20set%20with%20%24n%24%20rows%2C%20draw%20%24n%24%20rows%20**with%20replacement**.%20Some%20rows%20appear%20more%20than%20once%3B%20others%20are%20absent.%20Train%20one%20tree%20on%20this%20bootstrap%20sample.%0A%0A%20%20%20%20Repeat%20this%20independently%20for%20every%20tree.%20An%20omitted%20row%20is%20called%20out-of-bag%20for%20that%20tree.%0A%0A%20%20%20%20%23%23%23%202.%20Randomize%20each%20split%0A%0A%20%20%20%20At%20a%20node%2C%20do%20not%20inspect%20every%20feature.%20Draw%20a%20random%20subset%20and%20find%20the%20best%20split%20only%20inside%20that%20subset.%20This%20prevents%20one%20very%20strong%20feature%20from%20making%20all%20trees%20nearly%20identical.%0A%0A%20%20%20%20%23%23%23%203.%20Combine%20predictions%0A%0A%20%20%20%20For%20regression%20with%20%24B%24%20trees%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cboxed%7B%0A%20%20%20%20%5Chat%20y(x)%3D%5Cfrac1B%5Csum_%7Bb%3D1%7D%5E%7BB%7D%5Chat%20y_b(x)%0A%20%20%20%20%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20For%20classification%2C%20average%20class%20probabilities.%20If%20three%20trees%20predict%20class-1%20probabilities%0A%0A%20%20%20%20%24%24%0A%20%20%20%200.2%2C%5Cquad0.7%2C%5Cquad0.8%2C%0A%20%20%20%20%24%24%0A%0A%20%20%20%20the%20forest%20predicts%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Chat%20p%3D%5Cfrac%7B0.2%2B0.7%2B0.8%7D%7B3%7D%5Capprox0.57.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%23%23%23%20Why%20averaging%20helps%0A%0A%20%20%20%20If%20errors%20from%20different%20trees%20are%20not%20perfectly%20correlated%2C%20positive%20and%20negative%20deviations%20partly%20cancel.%20More%20trees%20reduce%20the%20randomness%20of%20the%20average%2C%20but%20they%20do%20not%20remove%20a%20bias%20shared%20by%20every%20tree.%0A%0A%20%20%20%20Out-of-bag%20predictions%20use%20only%20trees%20that%20did%20not%20train%20on%20a%20given%20row.%20They%20provide%20a%20useful%20internal%20estimate%2C%20but%20they%20do%20not%20replace%20a%20deployment-shaped%20validation%20design.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20A%20practical%20example%0A%0A%20%20%20%20For%20equipment%20failure%20classification%2C%20sensor%20summaries%2C%20machine%20type%2C%20age%2C%20and%20maintenance%20counts%20can%20interact%20in%20messy%20ways.%20Random%20Forest%20gives%20a%20strong%20baseline%20with%20little%20scaling%20and%20exposes%20whether%20nonlinear%20tabular%20structure%20is%20valuable.%0A%0A%20%20%20%20%23%23%20When%20to%20use%20it%0A%0A%20%20%20%20-%20You%20want%20a%20dependable%20nonlinear%20tabular%20baseline.%0A%20%20%20%20-%20Preprocessing%20time%20is%20limited.%0A%20%20%20%20-%20Thresholds%20and%20feature%20interactions%20are%20expected.%0A%20%20%20%20-%20Training%20can%20be%20parallelized.%0A%20%20%20%20-%20Some%20resistance%20to%20noise%20and%20hyperparameter%20mistakes%20is%20useful.%0A%0A%20%20%20%20%23%23%20When%20to%20avoid%20it%0A%0A%20%20%20%20-%20Smooth%20extrapolation%20beyond%20observed%20values%20matters.%0A%20%20%20%20-%20The%20model%20must%20be%20tiny%20or%20ultra-low-latency.%0A%20%20%20%20-%20A%20single%20simple%20explanation%20is%20required%20for%20each%20decision.%0A%20%20%20%20-%20Sparse%20extremely%20high-dimensional%20text%20is%20better%20matched%20by%20a%20linear%20model.%0A%0A%20%20%20%20%23%23%20Important%20controls%0A%0A%20%20%20%20%7C%20Control%20%7C%20Role%20%7C%0A%20%20%20%20%7C---------%7C------%7C%0A%20%20%20%20%7C%20%60n_estimators%60%20%7C%20More%20trees%20reduce%20Monte%20Carlo%20noise%20but%20increase%20cost%20%7C%0A%20%20%20%20%7C%20%60max_features%60%20%7C%20Lower%20values%20increase%20tree%20diversity%20%7C%0A%20%20%20%20%7C%20%60min_samples_leaf%60%20%7C%20Larger%20leaves%20smooth%20predictions%20%7C%0A%20%20%20%20%7C%20%60max_depth%60%20%7C%20Limits%20complexity%20%7C%0A%20%20%20%20%7C%20%60class_weight%60%20%7C%20Adjusts%20fitting%20emphasis%20for%20imbalance%20%7C%0A%0A%20%20%20%20Out-of-bag%20examples%20can%20provide%20an%20internal%20performance%20estimate%2C%20but%20a%20deployment-shaped%20validation%20set%20is%20still%20necessary.%0A%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20Notebook%20%E2%80%94%20averaging%20unstable%20trees%0A%0A%20%20%20%20**Question%3A**%20does%20averaging%20reduce%20sensitivity%20to%20the%20particular%20training%20sample%3F%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%20%20%20%20from%20sklearn.datasets%20import%20make_classification%0A%20%20%20%20from%20sklearn.ensemble%20import%20RandomForestClassifier%0A%20%20%20%20from%20sklearn.model_selection%20import%20train_test_split%0A%20%20%20%20from%20sklearn.tree%20import%20DecisionTreeClassifier%0A%0A%20%20%20%20X%2C%20y%20%3D%20make_classification(%0A%20%20%20%20%20%20%20%20n_samples%3D900%2C%20n_features%3D18%2C%20n_informative%3D8%2C%20flip_y%3D0.08%2C%20random_state%3D2%0A%20%20%20%20)%0A%20%20%20%20X_train%2C%20X_test%2C%20y_train%2C%20y_test%20%3D%20train_test_split(%0A%20%20%20%20%20%20%20%20X%2C%20y%2C%20test_size%3D0.4%2C%20stratify%3Dy%2C%20random_state%3D11%0A%20%20%20%20)%0A%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20DecisionTreeClassifier%2C%0A%20%20%20%20%20%20%20%20RandomForestClassifier%2C%0A%20%20%20%20%20%20%20%20X_test%2C%0A%20%20%20%20%20%20%20%20X_train%2C%0A%20%20%20%20%20%20%20%20np%2C%0A%20%20%20%20%20%20%20%20plt%2C%0A%20%20%20%20%20%20%20%20y_test%2C%0A%20%20%20%20%20%20%20%20y_train%2C%0A%20%20%20%20)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20We%20repeatedly%20change%20the%20bootstrap%20sample%20while%20leaving%20the%20task%20unchanged.%20Each%0A%20%20%20%20run%20fits%20one%20tree%20and%20one%20forest.%20The%20resulting%20score%20lists%20measure%20sensitivity%20to%0A%20%20%20%20the%20sampled%20data%2C%20which%20is%20the%20instability%20averaging%20is%20designed%20to%20reduce.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20DecisionTreeClassifier%2C%0A%20%20%20%20RandomForestClassifier%2C%0A%20%20%20%20X_test%2C%0A%20%20%20%20X_train%2C%0A%20%20%20%20np%2C%0A%20%20%20%20y_test%2C%0A%20%20%20%20y_train%2C%0A)%3A%0A%20%20%20%20tree_scores%2C%20forest_scores%20%3D%20%5B%5D%2C%20%5B%5D%0A%20%20%20%20for%20seed%20in%20range(20)%3A%0A%20%20%20%20%20%20%20%20rng%20%3D%20np.random.default_rng(seed)%0A%20%20%20%20%20%20%20%20sample%20%3D%20rng.choice(len(X_train)%2C%20len(X_train)%2C%20replace%3DTrue)%0A%20%20%20%20%20%20%20%20tree%20%3D%20DecisionTreeClassifier(random_state%3Dseed).fit(X_train%5Bsample%5D%2C%20y_train%5Bsample%5D)%0A%20%20%20%20%20%20%20%20forest%20%3D%20RandomForestClassifier(%0A%20%20%20%20%20%20%20%20%20%20%20%20n_estimators%3D150%2C%20min_samples_leaf%3D2%2C%20random_state%3Dseed%2C%20n_jobs%3D-1%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20forest.fit(X_train%5Bsample%5D%2C%20y_train%5Bsample%5D)%0A%20%20%20%20%20%20%20%20tree_scores.append(tree.score(X_test%2C%20y_test))%0A%20%20%20%20%20%20%20%20forest_scores.append(forest.score(X_test%2C%20y_test))%0A%0A%20%20%20%20print(f%22tree%3A%20%20%20mean%3D%7Bnp.mean(tree_scores)%3A.3f%7D%2C%20spread%3D%7Bnp.std(tree_scores)%3A.3f%7D%22)%0A%20%20%20%20print(f%22forest%3A%20mean%3D%7Bnp.mean(forest_scores)%3A.3f%7D%2C%20spread%3D%7Bnp.std(forest_scores)%3A.3f%7D%22)%0A%20%20%20%20return%20forest_scores%2C%20tree_scores%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20The%20two%20lists%20contain%20performance%20from%20matched%20resamples.%20Their%20distributions%20now%0A%20%20%20%20show%20both%20average%20quality%20and%20spread%3A%20a%20useful%20forest%20should%20usually%20be%20stronger%0A%20%20%20%20and%2C%20more%20importantly%20here%2C%20less%20variable%20than%20an%20individual%20tree.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(forest_scores%2C%20plt%2C%20tree_scores)%3A%0A%20%20%20%20plt.boxplot(%5Btree_scores%2C%20forest_scores%5D%2C%20tick_labels%3D%5B%22one%20tree%22%2C%20%22forest%22%5D)%0A%20%20%20%20plt.ylabel(%22test%20accuracy%22)%0A%20%20%20%20plt.title(%22Sensitivity%20to%20bootstrap%20samples%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20Evaluation%20and%20diagnosis%0A%0A%20%20%20%20Compare%20it%20with%20both%20a%20linear%20baseline%20and%20a%20boosted-tree%20model.%20Inspect%20learning%20curves%20to%20see%20whether%20more%20data%20may%20help.%20Use%20permutation%20importance%20or%20carefully%20interpreted%20SHAP%20values%20rather%20than%20relying%20only%20on%20impurity%20importance.%0A%0A%20%20%20%20Classification%20probabilities%20may%20need%20calibration.%20Rare%20categories%20and%20values%20outside%20the%20training%20range%20deserve%20explicit%20tests.%0A%0A%20%20%20%20%23%23%20Cost%20profile%0A%0A%20%20%20%20Training%20parallelizes%20well.%20Prediction%20cost%20grows%20with%20the%20number%20and%20depth%20of%20trees%3B%20model%20files%20can%20become%20large.%0A%0A%20%20%20%20%23%23%20Related%20models%0A%0A%20%20%20%20-%20**Decision%20Tree**%20is%20the%20readable%20but%20unstable%20base%20learner.%0A%20%20%20%20-%20Extra%20Trees%20adds%20more%20split%20randomization.%0A%20%20%20%20-%20**Gradient%20Boosting**%20builds%20trees%20sequentially%20and%20often%20achieves%20higher%20tabular%20accuracy%20with%20more%20tuning%20sensitivity.%0A%0A%20%20%20%20%23%23%20Why%20diversity%20matters%0A%0A%20%20%20%20Random%20Forest%20is%20an%20excellent%20%22get%20something%20strong%20and%20sane%20quickly%22%20model%20for%20ordinary%20tables.%0A%0A%20%20%20%20%23%23%20Concept%20references%0A%0A%20%20%20%20-%20%5BBias%20and%20Variance%5D(%2Fconcepts%2Fbias_variance)%20%E2%80%94%20why%20averaging%20unstable%20fits%20can%20help.%0A%20%20%20%20-%20%5BGeneralization%5D(%2Fconcepts%2Fgeneralization)%20%E2%80%94%20how%20to%20test%20whether%20that%20stability%20survives%20new%20data.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
8359eb64cbcc76eabad88c54645e9e55