import%20marimo%0A%0A__generated_with%20%3D%20%220.24.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20pandas%20as%20pd%0A%20%20%20%20import%20plotly.graph_objects%20as%20go%0A%20%20%20%20from%20plotly.subplots%20import%20make_subplots%0A%20%20%20%20from%20sklearn.linear_model%20import%20HuberRegressor%2C%20LinearRegression%0A%20%20%20%20from%20sklearn.metrics%20import%20mean_squared_error%2C%20r2_score%0A%0A%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20HuberRegressor%2C%0A%20%20%20%20%20%20%20%20LinearRegression%2C%0A%20%20%20%20%20%20%20%20go%2C%0A%20%20%20%20%20%20%20%20make_subplots%2C%0A%20%20%20%20%20%20%20%20mean_squared_error%2C%0A%20%20%20%20%20%20%20%20mo%2C%0A%20%20%20%20%20%20%20%20np%2C%0A%20%20%20%20%20%20%20%20pd%2C%0A%20%20%20%20%20%20%20%20r2_score%2C%0A%20%20%20%20)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%20Note%2033%3A%20Huber%20Loss%2C%20M-Estimation%2C%20and%20Robust%20Regression%20Dynamics%0A%0A%20%20%20%20%26larr%3B%20Previous%20Note%3A%20%5B32%20Elastic%20Net%5D(32_elastic_net.py)%20%7C%20Next%20Note%3A%20%5B34%20Mahalanobis%20Distance%5D(34_mahalanobis_distance.py)%20%26rarr%3B%0A%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20%5Ba%5D%20Why%20do%20you%20need%20to%20know%20these%20concepts%3F%0A%0A%20%20%20%20In%20real-world%20data%20science%20and%20deep%20learning%20systems%2C%20sensor%20anomalies%2C%20telemetry%20glitches%2C%20data%20entry%20errors%2C%20and%20financial%20market%20shocks%20create%20heavy-tailed%20noise%20and%20extreme%20outliers.%0A%0A%20%20%20%20The%20choice%20of%20loss%20function%20dictates%20how%20a%20model%20reacts%20to%20these%20anomalies%3A%0A%20%20%20%201.%20**The%20Vulnerability%20of%20Mean%20Squared%20Error%20(%24L_2%24%20Loss)**%3A%0A%20%20%20%20%20%20%20Under%20quadratic%20loss%20%24L(e)%20%3D%20%5Cfrac%7B1%7D%7B2%7D%20e%5E2%24%2C%20the%20gradient%20with%20respect%20to%20error%20is%20the%20error%20itself%20(%24%5Cfrac%7BdL%7D%7Bde%7D%20%3D%20e%24).%20A%20single%20extreme%20outlier%20with%20residual%20%24e%20%3D%20100%24%20exerts%20%24100%5Ctimes%24%20the%20gradient%20pull%20of%20a%20typical%20observation%20(%24e%20%3D%201%24)%2C%20exerting%20massive%20leverage%20that%20tilts%20the%20fitted%20hyperplane%20and%20corrupts%20parameter%20estimates.%0A%20%20%20%202.%20**The%20Pitfalls%20of%20Mean%20Absolute%20Error%20(%24L_1%24%20Loss)**%3A%0A%20%20%20%20%20%20%20Absolute%20loss%20%24L(e)%20%3D%20%7Ce%7C%24%20has%20a%20bounded%20derivative%20(%24%5Cpm%201%24)%20and%20resists%20outliers%2C%20but%20its%20gradient%20is%20discontinuous%20and%20non-differentiable%20at%20%24e%20%3D%200%24.%20Near%20the%20optimum%2C%20gradient%20descent%20oscillates%20perpetually%2C%20preventing%20smooth%20convergence.%0A%20%20%20%203.%20**Peter%20Huber's%20M-Estimation%20Breakthrough%20(1964)**%3A%0A%20%20%20%20%20%20%20**Huber%20Loss**%20combines%20the%20best%20properties%20of%20both%20regimes%3A%0A%20%20%20%20%20%20%20-%20**Quadratic%20near%20zero**%20(%24%7Ce%7C%20%5Cleq%20%5Cdelta%24)%3A%20Smooth%2C%20continuously%20differentiable%2C%20enabling%20rapid%20convergence%20without%20oscillation.%0A%20%20%20%20%20%20%20-%20**Linear%20in%20the%20tails**%20(%24%7Ce%7C%20%3E%20%5Cdelta%24)%3A%20Bounds%20the%20influence%20function%20to%20%24%5B-%5Cdelta%2C%20%5Cdelta%5D%24%2C%20neutralizing%20the%20leverage%20of%20severe%20outliers.%0A%20%20%20%204.%20**Standard%20in%20Modern%20Computer%20Vision**%3A%20In%20object%20detection%20frameworks%20(Fast%2FFaster%20R-CNN%2C%20YOLO%2C%20SSD)%2C%20bounding%20box%20regression%20uses%20**Smooth%20%24L_1%24%20Loss**%20(identically%20Huber%20loss%20with%20%24%5Cdelta%20%3D%201%24)%20to%20prevent%20gradient%20explosion%20from%20misaligned%20anchor%20proposals%20while%20achieving%20sub-pixel%20precision.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20%5Bb%5D%20Concept%20explanation%20with%20their%20role%20in%20ML%2FAI%2FStats%3F%0A%0A%20%20%20%20%23%23%23%201.%20Mathematical%20Formulation%0A%0A%20%20%20%20Let%20%24e%20%3D%20y%20-%20%5Chat%7By%7D%24%20denote%20the%20residual%20prediction%20error.%20The%20Huber%20loss%20with%20threshold%20parameter%20%24%5Cdelta%20%3E%200%24%20is%20defined%20as%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20L_%5Cdelta(e)%20%3D%20%5Cbegin%7Bcases%7D%0A%20%20%20%20%5Cfrac%7B1%7D%7B2%7D%20e%5E2%20%26%20%5Ctext%7Bif%20%7D%20%7Ce%7C%20%5Cleq%20%5Cdelta%20%5C%5C%0A%20%20%20%20%5Cdelta%20%7Ce%7C%20-%20%5Cfrac%7B1%7D%7B2%7D%20%5Cdelta%5E2%20%26%20%5Ctext%7Bif%20%7D%20%7Ce%7C%20%3E%20%5Cdelta%0A%20%20%20%20%5Cend%7Bcases%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20constant%20term%20%24-%5Cfrac%7B1%7D%7B2%7D%20%5Cdelta%5E2%24%20ensures%20%24%5Cmathcal%7BC%7D%5E1%24%20continuity%20(both%20the%20function%20value%20and%20its%20derivative%20match%20at%20the%20transition%20boundaries%20%24e%20%3D%20%5Cpm%20%5Cdelta%24).%0A%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%23%202.%20The%20Influence%20Function%20(Derivative)%0A%0A%20%20%20%20The%20derivative%20of%20the%20loss%20function%20with%20respect%20to%20the%20residual%2C%20known%20in%20robust%20statistics%20as%20the%20**influence%20function**%20%24%5Cpsi(e)%24%2C%20governs%20the%20gradient%20update%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cpsi_%5Cdelta(e)%20%3D%20%5Cfrac%7Bd%20L_%5Cdelta(e)%7D%7Bde%7D%20%3D%20%5Cbegin%7Bcases%7D%0A%20%20%20%20e%20%26%20%5Ctext%7Bif%20%7D%20%7Ce%7C%20%5Cleq%20%5Cdelta%20%5C%5C%0A%20%20%20%20%5Cdelta%20%5C%2C%20%5Coperatorname%7Bsign%7D(e)%20%26%20%5Ctext%7Bif%20%7D%20%7Ce%7C%20%3E%20%5Cdelta%0A%20%20%20%20%5Cend%7Bcases%7D%20%3D%20%5Coperatorname%7Bclip%7D(e%2C%20%5C%2C%20-%5Cdelta%2C%20%5C%2C%20%5Cdelta)%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%23%23%23%23%20Comparison%20of%20Influence%20Functions%3A%0A%20%20%20%20-%20**%24L_2%24%20Loss**%3A%20%24%5Cpsi(e)%20%3D%20e%20%5Cimplies%24%20Unbounded%20influence%20(%24%5Clim_%7B%7Ce%7C%20%5Cto%20%5Cinfty%7D%20%7C%5Cpsi(e)%7C%20%3D%20%5Cinfty%24).%0A%20%20%20%20-%20**%24L_1%24%20Loss**%3A%20%24%5Cpsi(e)%20%3D%20%5Coperatorname%7Bsign%7D(e)%20%5Cimplies%24%20Discontinuous%20jump%20at%20%24e%20%3D%200%24.%0A%20%20%20%20-%20**Huber%20Loss**%3A%20%24%5Cpsi_%5Cdelta(e)%20%3D%20%5Coperatorname%7Bclip%7D(e%2C%20-%5Cdelta%2C%20%5Cdelta)%20%5Cimplies%24%20Bounded%2C%20continuous%2C%20and%20strictly%20stable%20everywhere.%0A%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%23%203.%20Iteratively%20Reweighted%20Least%20Squares%20(IRLS)%0A%0A%20%20%20%20Minimizing%20the%20total%20Huber%20loss%20across%20%24n%24%20observations%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cmin_%7B%5Cboldsymbol%7B%5Cbeta%7D%7D%20%5Csum_%7Bi%3D1%7D%5En%20L_%5Cdelta(y_i%20-%20%5Cmathbf%7Bx%7D_i%5E%5Ctop%20%5Cboldsymbol%7B%5Cbeta%7D)%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Setting%20the%20gradient%20with%20respect%20to%20%24%5Cboldsymbol%7B%5Cbeta%7D%24%20to%20zero%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Csum_%7Bi%3D1%7D%5En%20%5Cpsi_%5Cdelta(e_i)%20%5Cmathbf%7Bx%7D_i%20%3D%20%5Cmathbf%7B0%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Rewriting%20%24%5Cpsi_%5Cdelta(e_i)%20%3D%20w_i%20e_i%24%2C%20where%20the%20Huber%20weights%20are%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20w_i%20%3D%20%5Cfrac%7B%5Cpsi_%5Cdelta(e_i)%7D%7Be_i%7D%20%3D%20%5Cbegin%7Bcases%7D%0A%20%20%20%201%20%26%20%5Ctext%7Bif%20%7D%20%7Ce_i%7C%20%5Cleq%20%5Cdelta%20%5C%5C%0A%20%20%20%20%5Cfrac%7B%5Cdelta%7D%7B%7Ce_i%7C%7D%20%26%20%5Ctext%7Bif%20%7D%20%7Ce_i%7C%20%3E%20%5Cdelta%0A%20%20%20%20%5Cend%7Bcases%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20This%20transforms%20the%20non-linear%20M-estimation%20problem%20into%20an%20equivalent%20**Weighted%20Least%20Squares%20(WLS)**%20problem%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BW%7D%5E%7B(t)%7D%20%5Cmathbf%7BX%7D%20%5Cboldsymbol%7B%5Cbeta%7D%5E%7B(t%2B1)%7D%20%3D%20%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BW%7D%5E%7B(t)%7D%20%5Cmathbf%7By%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20where%20%24%5Cmathbf%7BW%7D%20%3D%20%5Coperatorname%7Bdiag%7D(w_1%2C%20w_2%2C%20%5Cdots%2C%20w_n)%24.%0A%20%20%20%20-%20Observations%20with%20small%20residuals%20(%24%7Ce_i%7C%20%5Cleq%20%5Cdelta%24)%20receive%20full%20weight%20%24w_i%20%3D%201%24.%0A%20%20%20%20-%20Outliers%20(%24%7Ce_i%7C%20%5Cgg%20%5Cdelta%24)%20receive%20discounted%20weights%20%24w_i%20%5Cpropto%20%5Cfrac%7B1%7D%7B%7Ce_i%7C%7D%24%2C%20neutralizing%20their%20impact%20on%20the%20regression%20hyperplane.%0A%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%23%204.%20Selecting%20the%20Optimal%20Threshold%20%24%5Cdelta%24%0A%0A%20%20%20%20For%20Gaussian%20errors%20with%20standard%20deviation%20%24%5Csigma%24%2C%20setting%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cdelta%20%3D%201.345%20%5C%2C%20%5Csigma%0A%20%20%20%20%24%24%0A%0A%20%20%20%20yields%20**95%25%20asymptotic%20statistical%20efficiency**%20compared%20to%20OLS%20when%20the%20data%20is%20truly%20normal%2C%20while%20providing%20complete%20robustness%20against%20catastrophic%20outlier%20contamination.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(np%2C%20pd)%3A%0A%20%20%20%20%23%20Simulation%20Data%3A%201D%20Linear%20Relationship%20with%20Injected%20Severe%20Outliers%0A%20%20%20%20np.random.seed(47)%0A%20%20%20%20_n%20%3D%2060%0A%0A%20%20%20%20x_vals%20%3D%20np.sort(np.random.uniform(0.0%2C%2010.0%2C%20_n))%0A%20%20%20%20%23%20True%20relationship%3A%20y%20%3D%203.0%20%2B%202.5%20*%20x%20%2B%20Gaussian%20noise%0A%20%20%20%20_true_intercept%20%3D%203.0%0A%20%20%20%20_true_slope%20%3D%202.5%0A%20%20%20%20_noise%20%3D%20np.random.normal(0.0%2C%201.2%2C%20_n)%0A%20%20%20%20y_clean%20%3D%20_true_intercept%20%2B%20_true_slope%20*%20x_vals%20%2B%20_noise%0A%0A%20%20%20%20%23%20Incur%204%20catastrophic%20outliers%20(simulating%20sensor%20corruption)%0A%20%20%20%20y_contaminated%20%3D%20y_clean.copy()%0A%20%20%20%20_outlier_indices%20%3D%20%5B5%2C%2018%2C%2042%2C%2055%5D%0A%20%20%20%20y_contaminated%5B_outlier_indices%5B0%5D%5D%20%2B%3D%2030.0%0A%20%20%20%20y_contaminated%5B_outlier_indices%5B1%5D%5D%20-%3D%2025.0%0A%20%20%20%20y_contaminated%5B_outlier_indices%5B2%5D%5D%20-%3D%2035.0%0A%20%20%20%20y_contaminated%5B_outlier_indices%5B3%5D%5D%20%2B%3D%2028.0%0A%0A%20%20%20%20df_huber%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22x%22%3A%20x_vals%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22y%22%3A%20y_contaminated%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Is_Outlier%22%3A%20%5B1%20if%20i%20in%20_outlier_indices%20else%200%20for%20i%20in%20range(_n)%5D%2C%0A%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20)%0A%20%20%20%20return%20df_huber%2C%20x_vals%2C%20y_clean%2C%20y_contaminated%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20HuberRegressor%2C%0A%20%20%20%20LinearRegression%2C%0A%20%20%20%20df_huber%2C%0A%20%20%20%20go%2C%0A%20%20%20%20make_subplots%2C%0A%20%20%20%20np%2C%0A%20%20%20%20x_vals%2C%0A%20%20%20%20y_contaminated%2C%0A)%3A%0A%20%20%20%20%23%20Interactive%20Visualizations%20Cell%3A%0A%20%20%20%20%23%20Subplot%201%3A%20Loss%20Function%20and%20Derivative%20Curves%3A%20L2%20vs%20L1%20vs%20Huber%20(delta%20%3D%201.5)%0A%20%20%20%20%23%20Subplot%202%3A%20Robust%20Fit%20vs%20OLS%20on%20Contaminated%20Data%20(True%20vs%20OLS%20vs%20Huber)%0A%20%20%20%20%23%20Subplot%203%3A%20Huber%20Weight%20Decay%20Curve%20(w_i%20vs%20Residual%20%7Ce_i%7C)%0A%0A%20%20%20%20_fig%20%3D%20make_subplots(%0A%20%20%20%20%20%20%20%20rows%3D1%2C%0A%20%20%20%20%20%20%20%20cols%3D3%2C%0A%20%20%20%20%20%20%20%20subplot_titles%3D(%0A%20%20%20%20%20%20%20%20%20%20%20%20%221.%20Loss%20%26%20Influence%20Curves%20(L2%20vs.%20L1%20vs.%20Huber)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%222.%20Regression%20Fit%3A%20OLS%20vs.%20Huber%20under%20Outliers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%223.%20Huber%20Weight%20Discounting%20w(e)%20vs.%20%7Ce%7C%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20horizontal_spacing%3D0.09%2C%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Subplot%201%3A%20Loss%20Functions%20and%20Influence%20Functions%0A%20%20%20%20_e_range%20%3D%20np.linspace(-4.0%2C%204.0%2C%20200)%0A%20%20%20%20_delta%20%3D%201.5%0A%0A%20%20%20%20%23%20Losses%0A%20%20%20%20_loss_l2%20%3D%200.5%20*%20_e_range**2%0A%20%20%20%20_loss_l1%20%3D%20np.abs(_e_range)%0A%20%20%20%20_loss_huber%20%3D%20np.where(np.abs(_e_range)%20%3C%3D%20_delta%2C%200.5%20*%20_e_range**2%2C%20_delta%20*%20np.abs(_e_range)%20-%200.5%20*%20_delta**2)%0A%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D_e_range%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D_loss_l2%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%23ef4444%22%2C%20width%3D2%2C%20dash%3D%22dot%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22L2%20Loss%20(Quadratic)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D1%2C%0A%20%20%20%20)%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D_e_range%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D_loss_l1%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%23f59e0b%22%2C%20width%3D2%2C%20dash%3D%22dash%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22L1%20Loss%20(Absolute)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D1%2C%0A%20%20%20%20)%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D_e_range%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D_loss_huber%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%2310b981%22%2C%20width%3D3)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3Df%22Huber%20Loss%20(delta%3D%7B_delta%7D)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D1%2C%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Subplot%202%3A%20Fits%20on%20Contaminated%20Data%0A%20%20%20%20_X_mat%20%3D%20x_vals.reshape(-1%2C%201)%0A%0A%20%20%20%20%23%20OLS%20Model%0A%20%20%20%20_ols%20%3D%20LinearRegression().fit(_X_mat%2C%20y_contaminated)%0A%20%20%20%20_y_pred_ols%20%3D%20_ols.predict(_X_mat)%0A%0A%20%20%20%20%23%20Huber%20Model%0A%20%20%20%20_huber%20%3D%20HuberRegressor(epsilon%3D1.35).fit(_X_mat%2C%20y_contaminated)%0A%20%20%20%20_y_pred_huber%20%3D%20_huber.predict(_X_mat)%0A%0A%20%20%20%20_inliers%20%3D%20df_huber%5Bdf_huber%5B%22Is_Outlier%22%5D%20%3D%3D%200%5D%0A%20%20%20%20_outliers%20%3D%20df_huber%5Bdf_huber%5B%22Is_Outlier%22%5D%20%3D%3D%201%5D%0A%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D_inliers%5B%22x%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D_inliers%5B%22y%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22markers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20marker%3Ddict(size%3D6%2C%20color%3D%22%2364748b%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Inlier%20Data%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20hovertemplate%3D%22x%3A%20%25%7Bx%3A.2f%7D%3Cbr%3Ey%3A%20%25%7By%3A.2f%7D%3Cextra%3E%3C%2Fextra%3E%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D_outliers%5B%22x%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D_outliers%5B%22y%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22markers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20marker%3Ddict(size%3D10%2C%20symbol%3D%22x%22%2C%20color%3D%22%23ef4444%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Severe%20Outliers%20(%2B%2F-30)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20hovertemplate%3D%22Outlier%3A%20(%25%7Bx%3A.2f%7D%2C%20%25%7By%3A.2f%7D)%3Cextra%3E%3C%2Fextra%3E%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dx_vals%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D_y_pred_ols%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%23ef4444%22%2C%20width%3D2.5%2C%20dash%3D%22dash%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22OLS%20Fit%20(Tilted)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dx_vals%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D_y_pred_huber%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%2310b981%22%2C%20width%3D3)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Huber%20Fit%20(Robust)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20True%20line%3A%203%20%2B%202.5%20*%20x%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dx_vals%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D3.0%20%2B%202.5%20*%20x_vals%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%233b82f6%22%2C%20width%3D2%2C%20dash%3D%22dot%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Ground%20Truth%20(Clean)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Subplot%203%3A%20Huber%20Weights%20Decay%20Curve%0A%20%20%20%20_residuals_huber%20%3D%20np.abs(y_contaminated%20-%20_y_pred_huber)%0A%20%20%20%20_delta_eff%20%3D%20_huber.epsilon%20*%201.0%0A%20%20%20%20_weights_huber%20%3D%20np.where(_residuals_huber%20%3C%3D%20_delta_eff%2C%201.0%2C%20_delta_eff%20%2F%20_residuals_huber)%0A%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D_residuals_huber%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D_weights_huber%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22markers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20marker%3Ddict(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20size%3D8%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20color%3D%5B%22%23ef4444%22%20if%20w%20%3C%200.2%20else%20%22%2310b981%22%20for%20w%20in%20_weights_huber%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(width%3D1%2C%20color%3D%22%231e293b%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Sample%20Weights%20w_i%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20hovertemplate%3D%22Residual%3A%20%25%7Bx%3A.2f%7D%3Cbr%3EWeight%3A%20%25%7By%3A.3f%7D%3Cextra%3E%3C%2Fextra%3E%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D3%2C%0A%20%20%20%20)%0A%0A%20%20%20%20_fig.update_layout(%0A%20%20%20%20%20%20%20%20template%3D%22plotly_white%22%2C%0A%20%20%20%20%20%20%20%20height%3D480%2C%0A%20%20%20%20%20%20%20%20title%3Ddict(%0A%20%20%20%20%20%20%20%20%20%20%20%20text%3D%22Huber%20Robust%20Regression%3A%20M-Estimation%2C%20Outlier%20Invariance%2C%20and%20Weight%20Discounting%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D0.5%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20xanchor%3D%22center%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20font%3Ddict(size%3D16%2C%20family%3D%22Inter%2C%20system-ui%2C%20sans-serif%22)%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20legend%3Ddict(orientation%3D%22h%22%2C%20yanchor%3D%22bottom%22%2C%20y%3D-0.25%2C%20xanchor%3D%22center%22%2C%20x%3D0.5)%2C%0A%20%20%20%20%20%20%20%20margin%3Ddict(l%3D40%2C%20r%3D40%2C%20t%3D75%2C%20b%3D80)%2C%0A%20%20%20%20)%0A%0A%20%20%20%20_fig.update_xaxes(title_text%3D%22Residual%20Error%20(e)%22%2C%20row%3D1%2C%20col%3D1)%0A%20%20%20%20_fig.update_yaxes(title_text%3D%22Loss%20Value%20L(e)%22%2C%20range%3D%5B0%2C%208.5%5D%2C%20row%3D1%2C%20col%3D1)%0A%0A%20%20%20%20_fig.update_xaxes(title_text%3D%22Feature%20x%22%2C%20row%3D1%2C%20col%3D2)%0A%20%20%20%20_fig.update_yaxes(title_text%3D%22Target%20y%22%2C%20row%3D1%2C%20col%3D2)%0A%0A%20%20%20%20_fig.update_xaxes(title_text%3D%22Absolute%20Residual%20%7Ce_i%7C%22%2C%20row%3D1%2C%20col%3D3)%0A%20%20%20%20_fig.update_yaxes(title_text%3D%22Effective%20Huber%20Weight%20w_i%22%2C%20range%3D%5B-0.05%2C%201.05%5D%2C%20row%3D1%2C%20col%3D3)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20%5Bd%5D%20Code%20Examples%0A%0A%20%20%20%20Below%20we%20implement%20two%20rigorous%2C%20production-grade%20demonstrations%3A%0A%20%20%20%201.%20**Full%20Iteratively%20Reweighted%20Least%20Squares%20(IRLS)%20Solver%20from%20Scratch**%3A%20Pure%20NumPy%20implementation%20updating%20sample%20weights%20%24w_i%20%3D%20%5Cmin(1%2C%20%5Cdelta%20%2F%20%7Ce_i%7C)%24%20and%20solving%20weighted%20normal%20equations%20%24(%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BW%7D%20%5Cmathbf%7BX%7D)%5Cboldsymbol%7B%5Cbeta%7D%20%3D%20%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BW%7D%5Cmathbf%7By%7D%24%2C%20verified%20against%20Scikit-Learn%20%60HuberRegressor%60.%0A%20%20%20%202.%20**Contamination%20Robustness%20Benchmark**%3A%20Evaluating%20parameter%20recovery%20error%20against%20the%20ground-truth%20coefficients%20(%24%5Cbeta_0%20%3D%203.0%2C%20%5Cbeta_1%20%3D%202.5%24)%20for%20OLS%20vs.%20Huber%20regression%20as%20outlier%20contamination%20increases.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(HuberRegressor%2C%20LinearRegression%2C%20np%2C%20pd%2C%20x_vals%2C%20y_contaminated)%3A%0A%20%20%20%20%23%20Example%201%3A%20Pure%20NumPy%20Iteratively%20Reweighted%20Least%20Squares%20(IRLS)%20Huber%20Solver%0A%20%20%20%20_n%20%3D%20len(x_vals)%0A%20%20%20%20_X_aug%20%3D%20np.column_stack(%5Bnp.ones(_n)%2C%20x_vals%5D)%0A%20%20%20%20_y%20%3D%20y_contaminated%0A%20%20%20%20_delta%20%3D%202.5%0A%20%20%20%20_max_iter%20%3D%2050%0A%20%20%20%20_tol%20%3D%201e-6%0A%0A%20%20%20%20%23%20Initialize%20with%20OLS%20solution%0A%20%20%20%20_beta_irls%20%3D%20np.linalg.solve(_X_aug.T%20%40%20_X_aug%2C%20_X_aug.T%20%40%20_y)%0A%0A%20%20%20%20for%20_step%20in%20range(_max_iter)%3A%0A%20%20%20%20%20%20%20%20_res%20%3D%20_y%20-%20_X_aug%20%40%20_beta_irls%0A%20%20%20%20%20%20%20%20_abs_res%20%3D%20np.abs(_res)%0A%0A%20%20%20%20%20%20%20%20%23%20Compute%20Huber%20weights%0A%20%20%20%20%20%20%20%20_weights%20%3D%20np.where(_abs_res%20%3C%3D%20_delta%2C%201.0%2C%20_delta%20%2F%20np.maximum(_abs_res%2C%201e-8))%0A%20%20%20%20%20%20%20%20_W%20%3D%20np.diag(_weights)%0A%0A%20%20%20%20%20%20%20%20%23%20Solve%20weighted%20normal%20equations%3A%20(X%5ET%20W%20X)%20beta%20%3D%20X%5ET%20W%20y%0A%20%20%20%20%20%20%20%20_beta_next%20%3D%20np.linalg.solve(_X_aug.T%20%40%20_W%20%40%20_X_aug%2C%20_X_aug.T%20%40%20_W%20%40%20_y)%0A%0A%20%20%20%20%20%20%20%20if%20np.max(np.abs(_beta_next%20-%20_beta_irls))%20%3C%20_tol%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_beta_irls%20%3D%20_beta_next%0A%20%20%20%20%20%20%20%20%20%20%20%20break%0A%20%20%20%20%20%20%20%20_beta_irls%20%3D%20_beta_next%0A%0A%20%20%20%20%23%20Scikit-Learn%20Reference%0A%20%20%20%20_huber_sk%20%3D%20HuberRegressor(epsilon%3D1.35%2C%20max_iter%3D100).fit(x_vals.reshape(-1%2C%201)%2C%20_y)%0A%20%20%20%20_ols%20%3D%20LinearRegression().fit(x_vals.reshape(-1%2C%201)%2C%20_y)%0A%0A%20%20%20%20_df_irls_eval%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%22Model%22%3A%20%22Ground%20Truth%20Parameters%22%2C%20%22Intercept%20(beta_0)%22%3A%20%223.000%22%2C%20%22Slope%20(beta_1)%22%3A%20%222.500%22%2C%20%22Absolute%20Error%20vs%20Ground%20Truth%22%3A%20%220.000%20(Baseline)%22%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%22Model%22%3A%20%22OLS%20Regression%20(No%20Regularization)%22%2C%20%22Intercept%20(beta_0)%22%3A%20f%22%7B_ols.intercept_%3A.3f%7D%22%2C%20%22Slope%20(beta_1)%22%3A%20f%22%7B_ols.coef_%5B0%5D%3A.3f%7D%22%2C%20%22Absolute%20Error%20vs%20Ground%20Truth%22%3A%20f%22%7Bnp.abs(_ols.coef_%5B0%5D%20-%202.5)%3A.3f%7D%20(Severe%20Distortion)%22%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%22Model%22%3A%20%22IRLS%20Huber%20from%20Scratch%22%2C%20%22Intercept%20(beta_0)%22%3A%20f%22%7B_beta_irls%5B0%5D%3A.3f%7D%22%2C%20%22Slope%20(beta_1)%22%3A%20f%22%7B_beta_irls%5B1%5D%3A.3f%7D%22%2C%20%22Absolute%20Error%20vs%20Ground%20Truth%22%3A%20f%22%7Bnp.abs(_beta_irls%5B1%5D%20-%202.5)%3A.3f%7D%20(Near-Exact%20Recovery)%22%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%22Model%22%3A%20%22Scikit-Learn%20HuberRegressor%22%2C%20%22Intercept%20(beta_0)%22%3A%20f%22%7B_huber_sk.intercept_%3A.3f%7D%22%2C%20%22Slope%20(beta_1)%22%3A%20f%22%7B_huber_sk.coef_%5B0%5D%3A.3f%7D%22%2C%20%22Absolute%20Error%20vs%20Ground%20Truth%22%3A%20f%22%7Bnp.abs(_huber_sk.coef_%5B0%5D%20-%202.5)%3A.3f%7D%20(Robust%20Convergence)%22%7D%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20HuberRegressor%2C%0A%20%20%20%20LinearRegression%2C%0A%20%20%20%20mean_squared_error%2C%0A%20%20%20%20pd%2C%0A%20%20%20%20r2_score%2C%0A%20%20%20%20x_vals%2C%0A%20%20%20%20y_clean%2C%0A%20%20%20%20y_contaminated%2C%0A)%3A%0A%20%20%20%20%23%20Example%202%3A%20Clean%20Performance%20Evaluation%20on%20Uncontaminated%20Ground-Truth%0A%20%20%20%20%23%20How%20well%20do%20OLS%20and%20Huber%20predict%20the%20true%20underlying%20signal%20when%20trained%20on%20contaminated%20data%3F%0A%20%20%20%20_X_mat%20%3D%20x_vals.reshape(-1%2C%201)%0A%0A%20%20%20%20_ols%20%3D%20LinearRegression().fit(_X_mat%2C%20y_contaminated)%0A%20%20%20%20_huber%20%3D%20HuberRegressor().fit(_X_mat%2C%20y_contaminated)%0A%0A%20%20%20%20_pred_ols_clean%20%3D%20_ols.predict(_X_mat)%0A%20%20%20%20_pred_huber_clean%20%3D%20_huber.predict(_X_mat)%0A%0A%20%20%20%20%23%20Evaluate%20against%20pure%20uncontaminated%20target%20y_clean%0A%20%20%20%20_mse_ols_clean%20%3D%20mean_squared_error(y_clean%2C%20_pred_ols_clean)%0A%20%20%20%20_mse_huber_clean%20%3D%20mean_squared_error(y_clean%2C%20_pred_huber_clean)%0A%0A%20%20%20%20_r2_ols_clean%20%3D%20r2_score(y_clean%2C%20_pred_ols_clean)%0A%20%20%20%20_r2_huber_clean%20%3D%20r2_score(y_clean%2C%20_pred_huber_clean)%0A%0A%20%20%20%20_df_clean_eval%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Model%20Evaluated%22%3A%20%22Ordinary%20Least%20Squares%20(OLS)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22True%20Signal%20Test%20MSE%22%3A%20f%22%7B_mse_ols_clean%3A.3f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22True%20Signal%20Test%20R%5E2%22%3A%20f%22%7B_r2_ols_clean%3A.4f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Generalization%20Impact%22%3A%20%22Catastrophic%20degradation%20due%20to%20outlier%20leverage%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Model%20Evaluated%22%3A%20%22Huber%20Robust%20Regression%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22True%20Signal%20Test%20MSE%22%3A%20f%22%7B_mse_huber_clean%3A.3f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22True%20Signal%20Test%20R%5E2%22%3A%20f%22%7B_r2_huber_clean%3A.4f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Generalization%20Impact%22%3A%20%2297%25%2B%20of%20true%20signal%20preserved%20despite%20extreme%20anomalies%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
17c003ac2455a99eaf69db1fa9ea6fd9