import%20marimo%0A%0A__generated_with%20%3D%20%220.23.16%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%0A%20%20%20%20return%20(mo%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20METADATA%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22id%22%3A%20%22random_forest%22%2C%0A%20%20%20%20%20%20%20%20%22name%22%3A%20%22Random%20Forest%22%2C%0A%20%20%20%20%20%20%20%20%22types%22%3A%20%5B%22algorithm%22%5D%2C%0A%20%20%20%20%20%20%20%20%22families%22%3A%20%5B%22trees%22%2C%20%22ensembles%22%5D%2C%0A%20%20%20%20%20%20%20%20%22tasks%22%3A%20%5B%22classification%22%2C%20%22regression%22%5D%2C%0A%20%20%20%20%20%20%20%20%22data%22%3A%20%5B%22tabular%22%5D%2C%0A%20%20%20%20%20%20%20%20%22learning%22%3A%20%5B%22supervised%22%5D%2C%0A%20%20%20%20%20%20%20%20%22capacity%22%3A%20%22non_parametric%22%2C%0A%20%20%20%20%20%20%20%20%22mechanisms%22%3A%20%5B%22bagging%22%2C%20%22random_subspaces%22%2C%20%22threshold_rules%22%5D%2C%0A%20%20%20%20%20%20%20%20%22properties%22%3A%20%5B%22ensemble%22%2C%20%22nonlinear%22%5D%2C%0A%20%20%20%20%20%20%20%20%22constraints%22%3A%20%5B%22poor_extrapolation%22%5D%2C%0A%20%20%20%20%20%20%20%20%22difficulty%22%3A%20%22beginner%22%2C%0A%20%20%20%20%20%20%20%20%22status%22%3A%20%22complete%22%2C%0A%20%20%20%20%20%20%20%20%22explainability%22%3A%20%22medium%22%2C%0A%20%20%20%20%20%20%20%20%22training_cost%22%3A%20%22medium%22%2C%0A%20%20%20%20%20%20%20%20%22inference_cost%22%3A%20%22medium%22%2C%0A%20%20%20%20%20%20%20%20%22data_appetite%22%3A%20%22medium%22%2C%0A%20%20%20%20%7D%0A%0A%20%20%20%20mo.md(f%22%23%20%7BMETADATA%5B'name'%5D%7D%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20In%20one%20sentence%0A%0A%20%20%20%20Random%20Forest%20averages%20many%20deliberately%20varied%20decision%20trees%20to%20create%20a%20more%20stable%20predictor.%0A%0A%20%20%20%20%23%23%20Mental%20model%0A%0A%20%20%20%20Ask%20many%20competent%20but%20imperfect%20reviewers%20to%20judge%20slightly%20different%20evidence%2C%20then%20combine%20their%20answers.%20Averaging%20helps%20only%20when%20reviewers%20are%20not%20making%20exactly%20the%20same%20mistakes.%0A%0A%20%20%20%20Random%20Forest%20creates%20diversity%20by%20training%20each%20tree%20on%20a%20bootstrap%20sample%20and%20considering%20a%20random%20subset%20of%20features%20at%20each%20split.%0A%0A%20%20%20%20%23%23%20Input%20and%20output%0A%0A%20%20%20%20-%20**Input%3A**%20primarily%20tabular%20features.%0A%20%20%20%20-%20**Output%3A**%20averaged%20numeric%20prediction%20or%20class%20vote%2Fprobability.%0A%20%20%20%20-%20**Learns%3A**%20many%20decision%20trees.%0A%0A%20%20%20%20%23%23%20How%20it%20works%0A%0A%20%20%20%201.%20Sample%20training%20rows%20with%20replacement%20for%20each%20tree.%0A%20%20%20%202.%20Grow%20a%20tree%2C%20considering%20only%20a%20random%20feature%20subset%20at%20each%20split.%0A%20%20%20%203.%20Repeat%20independently.%0A%20%20%20%204.%20Average%20regression%20outputs%20or%20class%20probabilities.%0A%0A%20%20%20%20Deep%20individual%20trees%20have%20low%20bias%20but%20high%20variance.%20Averaging%20reduces%20variance%20while%20retaining%20nonlinear%20thresholds%20and%20interactions.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20A%20practical%20example%0A%0A%20%20%20%20For%20equipment%20failure%20classification%2C%20sensor%20summaries%2C%20machine%20type%2C%20age%2C%20and%20maintenance%20counts%20can%20interact%20in%20messy%20ways.%20Random%20Forest%20gives%20a%20strong%20baseline%20with%20little%20scaling%20and%20exposes%20whether%20nonlinear%20tabular%20structure%20is%20valuable.%0A%0A%20%20%20%20%23%23%20When%20to%20use%20it%0A%0A%20%20%20%20-%20You%20want%20a%20dependable%20nonlinear%20tabular%20baseline.%0A%20%20%20%20-%20Preprocessing%20time%20is%20limited.%0A%20%20%20%20-%20Thresholds%20and%20feature%20interactions%20are%20expected.%0A%20%20%20%20-%20Training%20can%20be%20parallelized.%0A%20%20%20%20-%20Some%20resistance%20to%20noise%20and%20hyperparameter%20mistakes%20is%20useful.%0A%0A%20%20%20%20%23%23%20When%20to%20avoid%20it%0A%0A%20%20%20%20-%20Smooth%20extrapolation%20beyond%20observed%20values%20matters.%0A%20%20%20%20-%20The%20model%20must%20be%20tiny%20or%20ultra-low-latency.%0A%20%20%20%20-%20A%20single%20simple%20explanation%20is%20required%20for%20each%20decision.%0A%20%20%20%20-%20Sparse%20extremely%20high-dimensional%20text%20is%20better%20matched%20by%20a%20linear%20model.%0A%0A%20%20%20%20%23%23%20Important%20controls%0A%0A%20%20%20%20%7C%20Control%20%7C%20Role%20%7C%0A%20%20%20%20%7C---------%7C------%7C%0A%20%20%20%20%7C%20%60n_estimators%60%20%7C%20More%20trees%20reduce%20Monte%20Carlo%20noise%20but%20increase%20cost%20%7C%0A%20%20%20%20%7C%20%60max_features%60%20%7C%20Lower%20values%20increase%20tree%20diversity%20%7C%0A%20%20%20%20%7C%20%60min_samples_leaf%60%20%7C%20Larger%20leaves%20smooth%20predictions%20%7C%0A%20%20%20%20%7C%20%60max_depth%60%20%7C%20Limits%20complexity%20%7C%0A%20%20%20%20%7C%20%60class_weight%60%20%7C%20Adjusts%20fitting%20emphasis%20for%20imbalance%20%7C%0A%0A%20%20%20%20Out-of-bag%20examples%20can%20provide%20an%20internal%20performance%20estimate%2C%20but%20a%20deployment-shaped%20validation%20set%20is%20still%20necessary.%0A%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20Notebook%20%E2%80%94%20averaging%20unstable%20trees%0A%0A%20%20%20%20**Question%3A**%20does%20averaging%20reduce%20sensitivity%20to%20the%20particular%20training%20sample%3F%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%20%20%20%20from%20sklearn.datasets%20import%20make_classification%0A%20%20%20%20from%20sklearn.ensemble%20import%20RandomForestClassifier%0A%20%20%20%20from%20sklearn.model_selection%20import%20train_test_split%0A%20%20%20%20from%20sklearn.tree%20import%20DecisionTreeClassifier%0A%0A%20%20%20%20X%2C%20y%20%3D%20make_classification(%0A%20%20%20%20%20%20%20%20n_samples%3D900%2C%20n_features%3D18%2C%20n_informative%3D8%2C%20flip_y%3D0.08%2C%20random_state%3D2%0A%20%20%20%20)%0A%20%20%20%20X_train%2C%20X_test%2C%20y_train%2C%20y_test%20%3D%20train_test_split(%0A%20%20%20%20%20%20%20%20X%2C%20y%2C%20test_size%3D0.4%2C%20stratify%3Dy%2C%20random_state%3D11%0A%20%20%20%20)%0A%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20DecisionTreeClassifier%2C%0A%20%20%20%20%20%20%20%20RandomForestClassifier%2C%0A%20%20%20%20%20%20%20%20X_test%2C%0A%20%20%20%20%20%20%20%20X_train%2C%0A%20%20%20%20%20%20%20%20np%2C%0A%20%20%20%20%20%20%20%20plt%2C%0A%20%20%20%20%20%20%20%20y_test%2C%0A%20%20%20%20%20%20%20%20y_train%2C%0A%20%20%20%20)%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20DecisionTreeClassifier%2C%0A%20%20%20%20RandomForestClassifier%2C%0A%20%20%20%20X_test%2C%0A%20%20%20%20X_train%2C%0A%20%20%20%20np%2C%0A%20%20%20%20y_test%2C%0A%20%20%20%20y_train%2C%0A)%3A%0A%20%20%20%20tree_scores%2C%20forest_scores%20%3D%20%5B%5D%2C%20%5B%5D%0A%20%20%20%20for%20seed%20in%20range(20)%3A%0A%20%20%20%20%20%20%20%20rng%20%3D%20np.random.default_rng(seed)%0A%20%20%20%20%20%20%20%20sample%20%3D%20rng.choice(len(X_train)%2C%20len(X_train)%2C%20replace%3DTrue)%0A%20%20%20%20%20%20%20%20tree%20%3D%20DecisionTreeClassifier(random_state%3Dseed).fit(X_train%5Bsample%5D%2C%20y_train%5Bsample%5D)%0A%20%20%20%20%20%20%20%20forest%20%3D%20RandomForestClassifier(%0A%20%20%20%20%20%20%20%20%20%20%20%20n_estimators%3D150%2C%20min_samples_leaf%3D2%2C%20random_state%3Dseed%2C%20n_jobs%3D-1%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20forest.fit(X_train%5Bsample%5D%2C%20y_train%5Bsample%5D)%0A%20%20%20%20%20%20%20%20tree_scores.append(tree.score(X_test%2C%20y_test))%0A%20%20%20%20%20%20%20%20forest_scores.append(forest.score(X_test%2C%20y_test))%0A%0A%20%20%20%20print(f%22tree%3A%20%20%20mean%3D%7Bnp.mean(tree_scores)%3A.3f%7D%2C%20spread%3D%7Bnp.std(tree_scores)%3A.3f%7D%22)%0A%20%20%20%20print(f%22forest%3A%20mean%3D%7Bnp.mean(forest_scores)%3A.3f%7D%2C%20spread%3D%7Bnp.std(forest_scores)%3A.3f%7D%22)%0A%20%20%20%20return%20forest_scores%2C%20tree_scores%0A%0A%0A%40app.cell%0Adef%20_(forest_scores%2C%20plt%2C%20tree_scores)%3A%0A%20%20%20%20plt.boxplot(%5Btree_scores%2C%20forest_scores%5D%2C%20tick_labels%3D%5B%22one%20tree%22%2C%20%22forest%22%5D)%0A%20%20%20%20plt.ylabel(%22test%20accuracy%22)%0A%20%20%20%20plt.title(%22Sensitivity%20to%20bootstrap%20samples%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20Evaluation%20and%20diagnosis%0A%0A%20%20%20%20Compare%20it%20with%20both%20a%20linear%20baseline%20and%20a%20boosted-tree%20model.%20Inspect%20learning%20curves%20to%20see%20whether%20more%20data%20may%20help.%20Use%20permutation%20importance%20or%20carefully%20interpreted%20SHAP%20values%20rather%20than%20relying%20only%20on%20impurity%20importance.%0A%0A%20%20%20%20Classification%20probabilities%20may%20need%20calibration.%20Rare%20categories%20and%20values%20outside%20the%20training%20range%20deserve%20explicit%20tests.%0A%0A%20%20%20%20%23%23%20Cost%20profile%0A%0A%20%20%20%20Training%20parallelizes%20well.%20Prediction%20cost%20grows%20with%20the%20number%20and%20depth%20of%20trees%3B%20model%20files%20can%20become%20large.%0A%0A%20%20%20%20%23%23%20Related%20models%0A%0A%20%20%20%20-%20**Decision%20Tree**%20is%20the%20readable%20but%20unstable%20base%20learner.%0A%20%20%20%20-%20Extra%20Trees%20adds%20more%20split%20randomization.%0A%20%20%20%20-%20**Gradient%20Boosting**%20builds%20trees%20sequentially%20and%20often%20achieves%20higher%20tabular%20accuracy%20with%20more%20tuning%20sensitivity.%0A%0A%20%20%20%20%23%23%20Practical%20takeaway%0A%0A%20%20%20%20Random%20Forest%20is%20an%20excellent%20%22get%20something%20strong%20and%20sane%20quickly%22%20model%20for%20ordinary%20tables.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
b58a9be0944721fb282e5d53d63e718c