import%20marimo%0A%0A__generated_with%20%3D%20%220.23.16%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%0A%20%20%20%20return%20(mo%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20METADATA%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22id%22%3A%20%22gradient_boosting%22%2C%0A%20%20%20%20%20%20%20%20%22name%22%3A%20%22Gradient%20Boosting%22%2C%0A%20%20%20%20%20%20%20%20%22types%22%3A%20%5B%22algorithm%22%5D%2C%0A%20%20%20%20%20%20%20%20%22families%22%3A%20%5B%22trees%22%2C%20%22ensembles%22%5D%2C%0A%20%20%20%20%20%20%20%20%22tasks%22%3A%20%5B%22classification%22%2C%20%22regression%22%5D%2C%0A%20%20%20%20%20%20%20%20%22data%22%3A%20%5B%22tabular%22%5D%2C%0A%20%20%20%20%20%20%20%20%22learning%22%3A%20%5B%22supervised%22%5D%2C%0A%20%20%20%20%20%20%20%20%22capacity%22%3A%20%22non_parametric%22%2C%0A%20%20%20%20%20%20%20%20%22mechanisms%22%3A%20%5B%22boosting%22%2C%20%22residual_fitting%22%2C%20%22threshold_rules%22%5D%2C%0A%20%20%20%20%20%20%20%20%22properties%22%3A%20%5B%22ensemble%22%2C%20%22nonlinear%22%5D%2C%0A%20%20%20%20%20%20%20%20%22constraints%22%3A%20%5B%22poor_extrapolation%22%2C%20%22sensitive_to_tuning%22%5D%2C%0A%20%20%20%20%20%20%20%20%22difficulty%22%3A%20%22intermediate%22%2C%0A%20%20%20%20%20%20%20%20%22status%22%3A%20%22complete%22%2C%0A%20%20%20%20%20%20%20%20%22explainability%22%3A%20%22medium%22%2C%0A%20%20%20%20%20%20%20%20%22training_cost%22%3A%20%22medium%22%2C%0A%20%20%20%20%20%20%20%20%22inference_cost%22%3A%20%22low%22%2C%0A%20%20%20%20%20%20%20%20%22data_appetite%22%3A%20%22medium%22%2C%0A%20%20%20%20%7D%0A%0A%20%20%20%20mo.md(f%22%23%20%7BMETADATA%5B'name'%5D%7D%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20In%20one%20sentence%0A%0A%20%20%20%20Gradient%20Boosting%20builds%20a%20sequence%20of%20small%20models%2C%20each%20correcting%20the%20current%20ensemble's%20mistakes.%0A%0A%20%20%20%20%23%23%20Mental%20model%0A%0A%20%20%20%20Write%20a%20rough%20draft%2C%20inspect%20its%20errors%2C%20add%20a%20targeted%20correction%2C%20and%20repeat.%20Unlike%20Random%20Forest%2C%20the%20learners%20are%20not%20independent%3A%20order%20is%20central.%0A%0A%20%20%20%20With%20decision%20trees%20as%20weak%20learners%2C%20each%20new%20tree%20follows%20the%20gradient%20of%20the%20chosen%20loss.%20Informally%2C%20it%20learns%20where%20and%20how%20the%20current%20predictions%20need%20to%20move.%0A%0A%20%20%20%20%23%23%20Input%20and%20output%0A%0A%20%20%20%20-%20**Input%3A**%20mainly%20tabular%20features.%0A%20%20%20%20-%20**Output%3A**%20numeric%20prediction%2C%20score%2C%20or%20class%20probability.%0A%20%20%20%20-%20**Learns%3A**%20an%20additive%20sequence%20of%20usually%20shallow%20trees.%0A%0A%20%20%20%20%23%23%20How%20it%20works%0A%0A%20%20%20%20The%20ensemble%20starts%20with%20a%20simple%20constant%20prediction.%20At%20each%20round%3A%0A%0A%20%20%20%201.%20Calculate%20how%20the%20loss%20would%20decrease%20if%20predictions%20changed.%0A%20%20%20%202.%20Fit%20a%20small%20tree%20to%20that%20correction%20signal.%0A%20%20%20%203.%20Add%20a%20scaled%20version%20of%20the%20tree%20to%20the%20ensemble.%0A%0A%20%20%20%20The%20learning%20rate%20controls%20the%20size%20of%20each%20correction.%20A%20smaller%20rate%20usually%20needs%20more%20trees%20and%20can%20generalize%20better%2C%20at%20higher%20training%20cost.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20A%20practical%20example%0A%0A%20%20%20%20For%20conversion%20prediction%2C%20many%20weak%20effects%20and%20interactions%20accumulate%3A%20traffic%20source%2C%20device%2C%20time%2C%20prior%20visits%2C%20price%2C%20and%20product%20category.%20Boosted%20trees%20can%20combine%20these%20efficiently%20without%20requiring%20a%20deep%20monolithic%20tree.%0A%0A%20%20%20%20%23%23%20When%20to%20use%20it%0A%0A%20%20%20%20-%20Predictive%20quality%20on%20structured%20tabular%20data%20is%20a%20priority.%0A%20%20%20%20-%20Nonlinear%20thresholds%20and%20interactions%20matter.%0A%20%20%20%20-%20You%20can%20afford%20careful%20validation%20and%20moderate%20tuning.%0A%20%20%20%20-%20Missing%20values%20and%20categorical%20features%20can%20be%20handled%20by%20the%20chosen%20implementation.%0A%20%20%20%20-%20Inference%20remains%20within%20the%20size%20and%20latency%20budget.%0A%0A%20%20%20%20%23%23%20When%20to%20avoid%20it%0A%0A%20%20%20%20-%20A%20simpler%20baseline%20already%20meets%20the%20product%20need.%0A%20%20%20%20-%20Labels%20are%20extremely%20noisy%20and%20boosting%20starts%20chasing%20residual%20noise.%0A%20%20%20%20-%20Smooth%20extrapolation%20is%20essential.%0A%20%20%20%20-%20Online%20updates%20are%20required%20but%20the%20selected%20implementation%20retrains%20in%20batches.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Important%20controls%0A%0A%20%20%20%20%7C%20Control%20%7C%20Role%20%7C%0A%20%20%20%20%7C---------%7C------%7C%0A%20%20%20%20%7C%20%60learning_rate%60%20%7C%20Contribution%20of%20each%20new%20tree%20%7C%0A%20%20%20%20%7C%20%60n_estimators%60%20%2F%20iterations%20%7C%20Number%20of%20correction%20rounds%20%7C%0A%20%20%20%20%7C%20Tree%20depth%20or%20leaf%20count%20%7C%20Complexity%20of%20each%20correction%20%7C%0A%20%20%20%20%7C%20%60subsample%60%20%7C%20Row%20sampling%20that%20can%20add%20regularization%20%7C%0A%20%20%20%20%7C%20Column%20sampling%20%7C%20Limits%20features%20used%20by%20each%20learner%20%7C%0A%20%20%20%20%7C%20Early%20stopping%20%7C%20Stops%20when%20validation%20improvement%20stalls%20%7C%0A%0A%20%20%20%20Tune%20learning%20rate%20and%20number%20of%20trees%20together.%20A%20large%20number%20of%20deep%20trees%20at%20a%20high%20learning%20rate%20is%20a%20common%20path%20to%20overfitting.%0A%0A%20%20%20%20%23%23%20Implementations%0A%0A%20%20%20%20The%20generic%20idea%20appears%20in%20scikit-learn%20GradientBoosting%2C%20histogram-based%20boosting%2C%20XGBoost%2C%20LightGBM%2C%20and%20CatBoost.%20They%20differ%20in%20tree%20construction%2C%20regularization%2C%20categorical%20handling%2C%20missing-value%20behavior%2C%20speed%2C%20and%20distributed%20support.%20Treat%20library%20choice%20as%20an%20engineering%20and%20validation%20decision%2C%20not%20a%20rebranding%20detail.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20Notebook%20%E2%80%94%20sequential%20correction%0A%0A%20%20%20%20**Question%3A**%20how%20do%20learning%20rate%20and%20iteration%20count%20work%20together%3F%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%20%20%20%20from%20sklearn.ensemble%20import%20GradientBoostingRegressor%0A%20%20%20%20from%20sklearn.metrics%20import%20mean_squared_error%0A%20%20%20%20from%20sklearn.model_selection%20import%20train_test_split%0A%0A%20%20%20%20rng%20%3D%20np.random.default_rng(5)%0A%20%20%20%20X%20%3D%20rng.uniform(-3%2C%203%2C%20size%3D(800%2C%201))%0A%20%20%20%20y%20%3D%20np.sin(2%20*%20X%5B%3A%2C%200%5D)%20%2B%200.25%20*%20X%5B%3A%2C%200%5D%20%2B%20rng.normal(0%2C%200.3%2C%20800)%0A%20%20%20%20X_train%2C%20X_test%2C%20y_train%2C%20y_test%20%3D%20train_test_split(X%2C%20y%2C%20test_size%3D0.4%2C%20random_state%3D5)%0A%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20GradientBoostingRegressor%2C%0A%20%20%20%20%20%20%20%20X_test%2C%0A%20%20%20%20%20%20%20%20X_train%2C%0A%20%20%20%20%20%20%20%20mean_squared_error%2C%0A%20%20%20%20%20%20%20%20np%2C%0A%20%20%20%20%20%20%20%20plt%2C%0A%20%20%20%20%20%20%20%20y_test%2C%0A%20%20%20%20%20%20%20%20y_train%2C%0A%20%20%20%20)%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20GradientBoostingRegressor%2C%0A%20%20%20%20X_test%2C%0A%20%20%20%20X_train%2C%0A%20%20%20%20mean_squared_error%2C%0A%20%20%20%20plt%2C%0A%20%20%20%20y_test%2C%0A%20%20%20%20y_train%2C%0A)%3A%0A%20%20%20%20fig%2C%20ax%20%3D%20plt.subplots(figsize%3D(8%2C%204))%0A%20%20%20%20for%20rate%20in%20%5B0.03%2C%200.1%2C%200.4%5D%3A%0A%20%20%20%20%20%20%20%20_model%20%3D%20GradientBoostingRegressor(%0A%20%20%20%20%20%20%20%20%20%20%20%20n_estimators%3D250%2C%20learning_rate%3Drate%2C%20max_depth%3D2%2C%20random_state%3D0%0A%20%20%20%20%20%20%20%20).fit(X_train%2C%20y_train)%0A%20%20%20%20%20%20%20%20errors%20%3D%20%5Bmean_squared_error(y_test%2C%20pred)%20for%20pred%20in%20_model.staged_predict(X_test)%5D%0A%20%20%20%20%20%20%20%20ax.plot(errors%2C%20label%3Df%22learning_rate%3D%7Brate%7D%22)%0A%20%20%20%20ax.set(xlabel%3D%22boosting%20iteration%22%2C%20ylabel%3D%22test%20MSE%22%2C%20title%3D%22Corrections%20accumulate%20over%20time%22)%0A%20%20%20%20ax.legend()%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(GradientBoostingRegressor%2C%20X_train%2C%20np%2C%20plt%2C%20y_train)%3A%0A%20%20%20%20_model%20%3D%20GradientBoostingRegressor(%0A%20%20%20%20%20%20%20%20n_estimators%3D120%2C%20learning_rate%3D0.05%2C%20max_depth%3D2%2C%20random_state%3D0%0A%20%20%20%20).fit(X_train%2C%20y_train)%0A%20%20%20%20grid%20%3D%20np.linspace(-3%2C%203%2C%20400).reshape(-1%2C%201)%0A%20%20%20%20plt.scatter(X_train%5B%3A%2C%200%5D%2C%20y_train%2C%20s%3D9%2C%20alpha%3D0.25%2C%20label%3D%22training%20data%22)%0A%20%20%20%20plt.plot(grid%5B%3A%2C%200%5D%2C%20_model.predict(grid)%2C%20color%3D%22crimson%22%2C%20linewidth%3D2%2C%20label%3D%22ensemble%22)%0A%20%20%20%20plt.legend()%0A%20%20%20%20plt.title(%22Many%20small%20trees%20form%20a%20smooth-looking%20fit%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20Evaluation%20and%20diagnosis%0A%0A%20%20%20%20Track%20training%20and%20validation%20loss%20by%20iteration.%20Use%20early%20stopping%20on%20a%20validation%20set%2C%20then%20evaluate%20once%20on%20untouched%20test%20data.%20Compare%20probability%20calibration%2C%20latency%2C%20and%20slice%20behavior%20%E2%80%94%20not%20only%20ranking%20metrics.%0A%0A%20%20%20%20Feature%20attributions%20help%20investigate%20behavior%20but%20do%20not%20establish%20causality.%20Stress-test%20missing%20values%20and%20unseen%20or%20rare%20categories.%0A%0A%20%20%20%20%23%23%20Cost%20profile%0A%0A%20%20%20%20Training%20is%20sequential%20across%20boosting%20rounds%20and%20less%20parallel%20than%20Random%20Forest.%20Histogram%20implementations%20are%20highly%20optimized.%20Inference%20cost%20grows%20with%20tree%20count%20and%20depth.%0A%0A%20%20%20%20%23%23%20Related%20models%0A%0A%20%20%20%20-%20**Decision%20Tree**%20supplies%20the%20common%20weak%20learner.%0A%20%20%20%20-%20**Random%20Forest**%20reduces%20variance%20through%20independent%20averaging.%0A%20%20%20%20-%20**Linear%20Regression**%20%2F%20**Logistic%20Regression**%20remains%20the%20essential%20simple%20baseline.%0A%0A%20%20%20%20%23%23%20Practical%20takeaway%0A%0A%20%20%20%20Gradient%20Boosting%20is%20often%20the%20model%20to%20beat%20on%20medium-sized%20tables%2C%20but%20its%20real%20advantage%20must%20survive%20fair%20splits%2C%20calibration%20checks%2C%20and%20operational%20cost.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
6a61140c740c52715c82ffe58d01fa8e