import%20marimo%0A%0A__generated_with%20%3D%20%220.24.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20pandas%20as%20pd%0A%20%20%20%20import%20plotly.graph_objects%20as%20go%0A%20%20%20%20from%20plotly.subplots%20import%20make_subplots%0A%20%20%20%20from%20sklearn.linear_model%20import%20LinearRegression%0A%0A%20%20%20%20return%20LinearRegression%2C%20go%2C%20make_subplots%2C%20mo%2C%20np%2C%20pd%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%20Note%2025%3A%20Predictive%20R-Squared%2C%20PRESS%20Residuals%2C%20and%20Leverage%20Diagnostics%0A%0A%20%20%20%20%26larr%3B%20Previous%20Note%3A%20%5B24%20Adjusted%20R-Squared%5D(24_adjusted_r_squared.py)%20%7C%20Next%20Note%3A%20%5B26%20Hotelling%20T-Squared%5D(26_hotelling.py)%20%26rarr%3B%0A%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20%5Ba%5D%20Why%20do%20you%20need%20to%20know%20these%20concepts%3F%0A%0A%20%20%20%20Both%20ordinary%20%24R%5E2%24%20and%20Adjusted%20%24R%5E2%24%20evaluate%20goodness-of-fit%20strictly%20on%20in-sample%20training%20data.%20A%20model%20can%20achieve%20an%20Adjusted%20%24R%5E2%24%20of%20%240.92%24%20yet%20fail%20catastrophically%20when%20deployed%20to%20predict%20new%2C%20unseen%20observations.%20This%20occurs%20when%20high-leverage%20training%20observations%20dictate%20the%20hyperplane%20slope%20or%20when%20polynomial%20basis%20expansions%20overfit%20local%20sample%20fluctuations.%0A%0A%20%20%20%20**Predictive%20%24R%5E2%24%20(%24R%5E2_%7B%5Ctext%7Bpred%7D%7D%24)**%20resolves%20this%20dilemma%20through%20Leave-One-Out%20Cross-Validation%20(LOOCV)%3A%0A%20%20%20%201.%20**Zero-Cost%20Out-of-Sample%20Evaluation**%3A%20Standard%20LOOCV%20requires%20training%20%24n%24%20separate%20models%2C%20which%20is%20computationally%20prohibitive%20for%20large%20datasets.%20In%20linear%20regression%2C%20the%20**Sherman-Morrison%20formula**%20enables%20the%20exact%20computation%20of%20all%20%24n%24%20leave-one-out%20residuals%20in%20a%20single%20matrix%20operation%20without%20refitting%20the%20model%20even%20once.%0A%20%20%20%202.%20**The%20PRESS%20Statistic**%3A%20The%20Prediction%20Error%20Sum%20of%20Squares%20(PRESS)%20aggregates%20the%20squared%20leave-one-out%20prediction%20errors%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Ctext%7BPRESS%7D%20%3D%20%5Csum_%7Bi%3D1%7D%5En%20%5Cleft(%5Cfrac%7Be_i%7D%7B1%20-%20h_%7Bii%7D%7D%5Cright)%5E2%0A%20%20%20%20%24%24%0A%0A%20%20%20%20where%20%24h_%7Bii%7D%24%20is%20the%20leverage%20of%20observation%20%24i%24.%20If%20a%20point%20has%20high%20leverage%20(%24h_%7Bii%7D%20%5Cto%201%24)%2C%20its%20residual%20is%20magnified%20by%20%24%5Cfrac%7B1%7D%7B(1%20-%20h_%7Bii%7D)%5E2%7D%24%2C%20heavily%20penalizing%20models%20that%20depend%20excessively%20on%20isolated%2C%20influential%20points.%0A%20%20%20%203.%20**Detecting%20Overfitting%20via%20the%20Generalization%20Gap**%3A%20Predictive%20%24R%5E2%24%20is%20defined%20as%20%241%20-%20%5Cfrac%7B%5Ctext%7BPRESS%7D%7D%7B%5Ctext%7BSS%7D_%7B%5Ctext%7Btot%7D%7D%7D%24.%20While%20raw%20%24R%5E2%24%20always%20increases%20with%20model%20complexity%2C%20Predictive%20%24R%5E2%24%20reaches%20a%20maximum%20and%20drops%20precipitously%20(often%20becoming%20strongly%20negative)%2C%20revealing%20precisely%20when%20additional%20features%20degrade%20generalization.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20%5Bb%5D%20Concept%20explanation%20with%20their%20role%20in%20ML%2FAI%2FStats%3F%0A%0A%20%20%20%20%23%23%23%201.%20The%20Hat%20(Projection)%20Matrix%20and%20Leverage%0A%0A%20%20%20%20In%20multiple%20linear%20regression%20with%20design%20matrix%20%24%5Cmathbf%7BX%7D%20%5Cin%20%5Cmathbb%7BR%7D%5E%7Bn%20%5Ctimes%20p%7D%24%20(including%20intercept)%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cmathbf%7By%7D%20%3D%20%5Cmathbf%7BX%7D%5Cboldsymbol%7B%5Cbeta%7D%20%2B%20%5Cboldsymbol%7B%5Cepsilon%7D%2C%20%5Cquad%20%5Chat%7B%5Cboldsymbol%7B%5Cbeta%7D%7D%20%3D%20(%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BX%7D)%5E%7B-1%7D%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7By%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20fitted%20values%20%24%5Chat%7B%5Cmathbf%7By%7D%7D%24%20are%20obtained%20via%20the%20orthogonal%20projection%20(Hat)%20matrix%20%24%5Cmathbf%7BH%7D%24%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Chat%7B%5Cmathbf%7By%7D%7D%20%3D%20%5Cmathbf%7BX%7D%5Chat%7B%5Cboldsymbol%7B%5Cbeta%7D%7D%20%3D%20%5Cmathbf%7BX%7D(%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BX%7D)%5E%7B-1%7D%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7By%7D%20%3D%20%5Cmathbf%7BH%7D%5Cmathbf%7By%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20diagonal%20elements%20%24h_%7Bii%7D%20%3D%20%5B%5Cmathbf%7BH%7D%5D_%7Bii%7D%24%20are%20the%20**leverage%20values**%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20h_%7Bii%7D%20%3D%20%5Cmathbf%7Bx%7D_i%5E%5Ctop%20(%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BX%7D)%5E%7B-1%7D%20%5Cmathbf%7Bx%7D_i%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%23%23%23%23%20Fundamental%20Properties%20of%20Leverage%3A%0A%20%20%20%20-%20Bounded%20in%20%24%5B0%2C%201%5D%24%3A%20%240%20%5Cleq%20h_%7Bii%7D%20%5Cleq%201%24.%0A%20%20%20%20-%20Sum%20of%20leverages%20equals%20the%20number%20of%20parameters%3A%20%24%5Coperatorname%7Btr%7D(%5Cmathbf%7BH%7D)%20%3D%20%5Csum_%7Bi%3D1%7D%5En%20h_%7Bii%7D%20%3D%20p%24.%0A%20%20%20%20-%20Average%20leverage%3A%20%24%5Cbar%7Bh%7D%20%3D%20%5Cfrac%7Bp%7D%7Bn%7D%24.%0A%20%20%20%20-%20High%20leverage%20threshold%3A%20An%20observation%20is%20deemed%20high-leverage%20if%20%24h_%7Bii%7D%20%3E%20%5Cfrac%7B2p%7D%7Bn%7D%24%20(or%20%24%5Cfrac%7B3p%7D%7Bn%7D%24).%0A%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%23%202.%20The%20PRESS%20Shortcut%20Derivation%20(Sherman-Morrison)%0A%0A%20%20%20%20Let%20%24(i)%24%20denote%20estimation%20with%20the%20%24i%24-th%20observation%20removed.%20The%20leave-one-out%20parameter%20estimate%20is%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Chat%7B%5Cboldsymbol%7B%5Cbeta%7D%7D_%7B(i)%7D%20%3D%20%5Cleft(%5Cmathbf%7BX%7D_%7B(i)%7D%5E%5Ctop%20%5Cmathbf%7BX%7D_%7B(i)%7D%5Cright)%5E%7B-1%7D%20%5Cmathbf%7BX%7D_%7B(i)%7D%5E%5Ctop%20%5Cmathbf%7By%7D_%7B(i)%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Notice%20that%20removing%20row%20%24%5Cmathbf%7Bx%7D_i%24%20is%20a%20symmetric%20rank-one%20downdate%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cmathbf%7BX%7D_%7B(i)%7D%5E%5Ctop%20%5Cmathbf%7BX%7D_%7B(i)%7D%20%3D%20%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BX%7D%20-%20%5Cmathbf%7Bx%7D_i%20%5Cmathbf%7Bx%7D_i%5E%5Ctop%0A%20%20%20%20%24%24%0A%0A%20%20%20%20By%20the%20Sherman-Morrison%20rank-one%20inverse%20formula%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Cleft(%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BX%7D%20-%20%5Cmathbf%7Bx%7D_i%20%5Cmathbf%7Bx%7D_i%5E%5Ctop%5Cright)%5E%7B-1%7D%20%3D%20(%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BX%7D)%5E%7B-1%7D%20%2B%20%5Cfrac%7B(%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BX%7D)%5E%7B-1%7D%20%5Cmathbf%7Bx%7D_i%20%5Cmathbf%7Bx%7D_i%5E%5Ctop%20(%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BX%7D)%5E%7B-1%7D%7D%7B1%20-%20h_%7Bii%7D%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Multiplying%20by%20%24%5Cmathbf%7BX%7D_%7B(i)%7D%5E%5Ctop%20%5Cmathbf%7By%7D_%7B(i)%7D%20%3D%20%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7By%7D%20-%20%5Cmathbf%7Bx%7D_i%20y_i%24%20and%20simplifying%20yields%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Chat%7B%5Cboldsymbol%7B%5Cbeta%7D%7D_%7B(i)%7D%20%3D%20%5Chat%7B%5Cboldsymbol%7B%5Cbeta%7D%7D%20-%20%5Cfrac%7B(%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BX%7D)%5E%7B-1%7D%5Cmathbf%7Bx%7D_i%20e_i%7D%7B1%20-%20h_%7Bii%7D%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20predicted%20value%20for%20the%20held-out%20point%20is%20%24%5Chat%7By%7D_%7B(i)%7D%20%3D%20%5Cmathbf%7Bx%7D_i%5E%5Ctop%20%5Chat%7B%5Cboldsymbol%7B%5Cbeta%7D%7D_%7B(i)%7D%24%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Chat%7By%7D_%7B(i)%7D%20%3D%20%5Cmathbf%7Bx%7D_i%5E%5Ctop%20%5Chat%7B%5Cboldsymbol%7B%5Cbeta%7D%7D%20-%20%5Cfrac%7B%5Cmathbf%7Bx%7D_i%5E%5Ctop%20(%5Cmathbf%7BX%7D%5E%5Ctop%20%5Cmathbf%7BX%7D)%5E%7B-1%7D%5Cmathbf%7Bx%7D_i%20e_i%7D%7B1%20-%20h_%7Bii%7D%7D%20%3D%20%5Chat%7By%7D_i%20-%20%5Cfrac%7Bh_%7Bii%7D%20e_i%7D%7B1%20-%20h_%7Bii%7D%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20Subtracting%20this%20from%20the%20observed%20target%20%24y_i%24%20yields%20the%20leave-one-out%20error%20%24e_%7B(i)%7D%24%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20e_%7B(i)%7D%20%3D%20y_i%20-%20%5Chat%7By%7D_%7B(i)%7D%20%3D%20y_i%20-%20%5Chat%7By%7D_i%20%2B%20%5Cfrac%7Bh_%7Bii%7D%20e_i%7D%7B1%20-%20h_%7Bii%7D%7D%20%3D%20e_i%20%5Cleft(1%20%2B%20%5Cfrac%7Bh_%7Bii%7D%7D%7B1%20-%20h_%7Bii%7D%7D%5Cright)%20%3D%20%5Cfrac%7Be_i%7D%7B1%20-%20h_%7Bii%7D%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20This%20identity%20proves%20that%20**Leave-One-Out%20residuals%20can%20be%20computed%20directly%20from%20standard%20OLS%20residuals%20and%20leverage%20values%20with%20zero%20re-fitting**.%0A%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%23%203.%20The%20PRESS%20Statistic%20and%20Predictive%20%24R%5E2%24%0A%0A%20%20%20%20The%20Prediction%20Error%20Sum%20of%20Squares%20(PRESS)%20is%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20%5Ctext%7BPRESS%7D%20%3D%20%5Csum_%7Bi%3D1%7D%5En%20e_%7B(i)%7D%5E2%20%3D%20%5Csum_%7Bi%3D1%7D%5En%20%5Cleft(%5Cfrac%7Be_i%7D%7B1%20-%20h_%7Bii%7D%7D%5Cright)%5E2%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20**Predictive%20%24R%5E2%24**%20is%20defined%20as%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20R%5E2_%7B%5Ctext%7Bpred%7D%7D%20%3D%201%20-%20%5Cfrac%7B%5Ctext%7BPRESS%7D%7D%7B%5Ctext%7BSS%7D_%7B%5Ctext%7Btot%7D%7D%7D%0A%20%20%20%20%24%24%0A%0A%20%20%20%20%23%23%23%23%20Comparison%20of%20the%20Three%20%24R%5E2%24%20Metrics%3A%0A%20%20%20%20%24%24%0A%20%20%20%20R%5E2%20%5Cgeq%20R%5E2_%7B%5Ctext%7Badj%7D%7D%20%5Cgeq%20R%5E2_%7B%5Ctext%7Bpred%7D%7D%0A%20%20%20%20%24%24%0A%20%20%20%20-%20**Raw%20%24R%5E2%24**%3A%20Evaluates%20in-sample%20fitting%20accuracy.%0A%20%20%20%20-%20**Adjusted%20%24R%5E2%24**%3A%20Penalizes%20degrees%20of%20freedom%20consumed%20by%20predictors.%0A%20%20%20%20-%20**Predictive%20%24R%5E2%24**%3A%20Evaluates%20true%20out-of-sample%20leave-one-out%20generalization.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(np%2C%20pd)%3A%0A%20%20%20%20%23%20Simulation%20Data%3A%20Non-linear%20relationship%20y%20%3D%201.5%20*%20x%20-%202.0%20*%20x%5E2%20%2B%200.8%20*%20x%5E3%20%2B%20noise%0A%20%20%20%20%23%20Designed%20to%20test%20polynomial%20models%20degree%201%20through%206%0A%20%20%20%20np.random.seed(47)%0A%20%20%20%20_n%20%3D%2035%0A%0A%20%20%20%20x_raw%20%3D%20np.sort(np.random.uniform(-1.8%2C%201.8%2C%20_n))%0A%20%20%20%20_true_signal%20%3D%201.5%20*%20x_raw%20-%202.0%20*%20(x_raw**2)%20%2B%200.8%20*%20(x_raw**3)%0A%20%20%20%20_noise%20%3D%20np.random.normal(0.0%2C%200.8%2C%20_n)%0A%0A%20%20%20%20%23%20Incur%20an%20intentional%20isolated%20high-leverage%20point%20at%20the%20extreme%20right%0A%20%20%20%20y_raw%20%3D%20_true_signal%20%2B%20_noise%0A%20%20%20%20y_raw%5B-1%5D%20%2B%3D%202.5%20%20%23%20high-leverage%20perturbed%20observation%0A%0A%20%20%20%20df_poly%20%3D%20pd.DataFrame(%7B%22x%22%3A%20np.round(x_raw%2C%203)%2C%20%22y%22%3A%20np.round(y_raw%2C%203)%7D)%0A%20%20%20%20return%20df_poly%2C%20x_raw%2C%20y_raw%0A%0A%0A%40app.cell%0Adef%20_(df_poly%2C%20go%2C%20make_subplots%2C%20np%2C%20x_raw%2C%20y_raw)%3A%0A%20%20%20%20%23%20Interactive%20Visualizations%20Cell%3A%0A%20%20%20%20%23%20Subplot%201%3A%20Fitted%20Polynomial%20curves%20(Degree%201%20Underfitting%2C%20Degree%203%20Optimal%2C%20Degree%206%20Overfitting)%0A%20%20%20%20%23%20Subplot%202%3A%20R%5E2%20vs.%20Adjusted%20R%5E2%20vs.%20Predictive%20R%5E2%20across%20Polynomial%20Degrees%201%20to%206%0A%20%20%20%20%23%20Subplot%203%3A%20Leverage%20(h_ii)%20vs.%20PRESS%20Residual%20Inflation%20Factor%201%20%2F%20(1%20-%20h_ii)%0A%0A%20%20%20%20_fig%20%3D%20make_subplots(%0A%20%20%20%20%20%20%20%20rows%3D1%2C%0A%20%20%20%20%20%20%20%20cols%3D3%2C%0A%20%20%20%20%20%20%20%20subplot_titles%3D(%0A%20%20%20%20%20%20%20%20%20%20%20%20%221.%20Polynomial%20Fits%3A%20Underfit%20vs.%20Optimal%20vs.%20Overfit%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%222.%20Generalization%20Gap%3A%20R%5E2%20vs.%20Adj%20R%5E2%20vs.%20Pred%20R%5E2%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%223.%20Leverage%20%26%20PRESS%20Residual%20Inflation%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20horizontal_spacing%3D0.09%2C%0A%20%20%20%20)%0A%0A%20%20%20%20_n%20%3D%20len(x_raw)%0A%20%20%20%20_ss_tot%20%3D%20np.sum((y_raw%20-%20np.mean(y_raw))%20**%202)%0A%0A%20%20%20%20%23%20Fit%20Polynomial%20Degrees%201%20through%206%0A%20%20%20%20_degrees%20%3D%20np.arange(1%2C%207)%0A%20%20%20%20_r2_list%20%3D%20%5B%5D%0A%20%20%20%20_adj_r2_list%20%3D%20%5B%5D%0A%20%20%20%20_pred_r2_list%20%3D%20%5B%5D%0A%0A%20%20%20%20_fits_to_plot%20%3D%20%7B%7D%0A%20%20%20%20_x_dense%20%3D%20np.linspace(x_raw.min()%2C%20x_raw.max()%2C%20100)%0A%0A%20%20%20%20for%20_d%20in%20_degrees%3A%0A%20%20%20%20%20%20%20%20_X_mat%20%3D%20np.vander(x_raw%2C%20_d%20%2B%201)%20%20%23%20includes%20column%20of%20ones%0A%20%20%20%20%20%20%20%20_beta%2C%20_%2C%20_%2C%20_%20%3D%20np.linalg.lstsq(_X_mat%2C%20y_raw%2C%20rcond%3DNone)%0A%20%20%20%20%20%20%20%20_y_hat%20%3D%20_X_mat%20%40%20_beta%0A%20%20%20%20%20%20%20%20_e%20%3D%20y_raw%20-%20_y_hat%0A%0A%20%20%20%20%20%20%20%20%23%20Hat%20matrix%0A%20%20%20%20%20%20%20%20_H%20%3D%20_X_mat%20%40%20np.linalg.solve(_X_mat.T%20%40%20_X_mat%2C%20_X_mat.T)%0A%20%20%20%20%20%20%20%20_h%20%3D%20np.diag(_H)%0A%0A%20%20%20%20%20%20%20%20_ss_res%20%3D%20np.sum(_e**2)%0A%20%20%20%20%20%20%20%20_press%20%3D%20np.sum((_e%20%2F%20(1.0%20-%20_h))%20**%202)%0A%0A%20%20%20%20%20%20%20%20_p%20%3D%20_d%20%2B%201%0A%20%20%20%20%20%20%20%20_r2%20%3D%201.0%20-%20(_ss_res%20%2F%20_ss_tot)%0A%20%20%20%20%20%20%20%20_adj_r2%20%3D%201.0%20-%20(1.0%20-%20_r2)%20*%20(_n%20-%201)%20%2F%20(_n%20-%20_p)%0A%20%20%20%20%20%20%20%20_pred_r2%20%3D%201.0%20-%20(_press%20%2F%20_ss_tot)%0A%0A%20%20%20%20%20%20%20%20_r2_list.append(_r2)%0A%20%20%20%20%20%20%20%20_adj_r2_list.append(_adj_r2)%0A%20%20%20%20%20%20%20%20_pred_r2_list.append(_pred_r2)%0A%0A%20%20%20%20%20%20%20%20if%20_d%20in%20%5B1%2C%203%2C%206%5D%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_X_dense%20%3D%20np.vander(_x_dense%2C%20_d%20%2B%201)%0A%20%20%20%20%20%20%20%20%20%20%20%20_fits_to_plot%5B_d%5D%20%3D%20_X_dense%20%40%20_beta%0A%0A%20%20%20%20%23%20Subplot%201%3A%20Fits%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Ddf_poly%5B%22x%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Ddf_poly%5B%22y%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22markers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20marker%3Ddict(size%3D7%2C%20color%3D%22%23334155%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Observed%20Samples%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D1%2C%0A%20%20%20%20)%0A%0A%20%20%20%20_fit_colors%20%3D%20%7B1%3A%20%22%23ef4444%22%2C%203%3A%20%22%2310b981%22%2C%206%3A%20%22%238b5cf6%22%7D%0A%20%20%20%20_fit_labels%20%3D%20%7B1%3A%20%22Degree%201%20(Underfit)%22%2C%203%3A%20%22Degree%203%20(Optimal)%22%2C%206%3A%20%22Degree%206%20(Overfit)%22%7D%0A%20%20%20%20for%20_d%2C%20_col%20in%20_fit_colors.items()%3A%0A%20%20%20%20%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20x%3D_x_dense%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20y%3D_fits_to_plot%5B_d%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D_col%2C%20width%3D2.5)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20name%3D_fit_labels%5B_d%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20col%3D1%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%23%20Subplot%202%3A%20Metric%20Trajectories%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D_degrees%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D_r2_list%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%2Bmarkers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%233b82f6%22%2C%20width%3D2.5)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Raw%20R%5E2%20(In-sample)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D_degrees%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D_adj_r2_list%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%2Bmarkers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%2310b981%22%2C%20width%3D2.5)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Adjusted%20R%5E2%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D_degrees%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D_pred_r2_list%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%2Bmarkers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%23ef4444%22%2C%20width%3D3%2C%20dash%3D%22dash%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Predictive%20R%5E2%20(PRESS)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Subplot%203%3A%20Leverage%20Diagnostic%20for%20Degree%203%0A%20%20%20%20_X_deg3%20%3D%20np.vander(x_raw%2C%204)%0A%20%20%20%20_H_deg3%20%3D%20_X_deg3%20%40%20np.linalg.solve(_X_deg3.T%20%40%20_X_deg3%2C%20_X_deg3.T)%0A%20%20%20%20_h_vals%20%3D%20np.diag(_H_deg3)%0A%20%20%20%20_inflation_factor%20%3D%201.0%20%2F%20(1.0%20-%20_h_vals)%0A%20%20%20%20_high_lev_thresh%20%3D%202.0%20*%204%20%2F%20_n%0A%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D_h_vals%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D_inflation_factor%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22markers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20marker%3Ddict(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20size%3D9%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20color%3D%5B%22%23ef4444%22%20if%20h%20%3E%20_high_lev_thresh%20else%20%22%233b82f6%22%20for%20h%20in%20_h_vals%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(width%3D1%2C%20color%3D%22%231e293b%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Observations%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20hovertemplate%3D%22Leverage%20h_ii%3A%20%25%7Bx%3A.3f%7D%3Cbr%3EPRESS%20Multiplier%3A%20%25%7By%3A.2f%7Dx%3Cextra%3E%3C%2Fextra%3E%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D3%2C%0A%20%20%20%20)%0A%0A%20%20%20%20_fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D%5B_high_lev_thresh%2C%20_high_lev_thresh%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D%5B1.0%2C%20_inflation_factor.max()%20*%201.05%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%23ef4444%22%2C%20dash%3D%22dash%22%2C%20width%3D1.5)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Threshold%202p%2Fn%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D3%2C%0A%20%20%20%20)%0A%0A%20%20%20%20_fig.update_layout(%0A%20%20%20%20%20%20%20%20template%3D%22plotly_white%22%2C%0A%20%20%20%20%20%20%20%20height%3D480%2C%0A%20%20%20%20%20%20%20%20title%3Ddict(%0A%20%20%20%20%20%20%20%20%20%20%20%20text%3D%22Predictive%20R%5E2%20%26%20PRESS%3A%20Diagnosing%20Overfitting%20and%20High-Leverage%20Outliers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D0.5%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20xanchor%3D%22center%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20font%3Ddict(size%3D16%2C%20family%3D%22Inter%2C%20system-ui%2C%20sans-serif%22)%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20legend%3Ddict(orientation%3D%22h%22%2C%20yanchor%3D%22bottom%22%2C%20y%3D-0.25%2C%20xanchor%3D%22center%22%2C%20x%3D0.5)%2C%0A%20%20%20%20%20%20%20%20margin%3Ddict(l%3D40%2C%20r%3D40%2C%20t%3D75%2C%20b%3D80)%2C%0A%20%20%20%20)%0A%0A%20%20%20%20_fig.update_xaxes(title_text%3D%22x%22%2C%20row%3D1%2C%20col%3D1)%0A%20%20%20%20_fig.update_yaxes(title_text%3D%22y%22%2C%20row%3D1%2C%20col%3D1)%0A%0A%20%20%20%20_fig.update_xaxes(title_text%3D%22Polynomial%20Degree%22%2C%20tickvals%3D_degrees%2C%20row%3D1%2C%20col%3D2)%0A%20%20%20%20_fig.update_yaxes(title_text%3D%22Metric%20Score%22%2C%20range%3D%5B-0.4%2C%201.05%5D%2C%20row%3D1%2C%20col%3D2)%0A%0A%20%20%20%20_fig.update_xaxes(title_text%3D%22Leverage%20(h_ii)%22%2C%20row%3D1%2C%20col%3D3)%0A%20%20%20%20_fig.update_yaxes(title_text%3D%22Error%20Inflation%201%20%2F%20(1%20-%20h_ii)%22%2C%20row%3D1%2C%20col%3D3)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20%5Bd%5D%20Code%20Examples%0A%0A%20%20%20%20Below%20we%20implement%20two%20end-to-end%20production%20algorithmic%20workflows%3A%0A%20%20%20%201.%20**Exact%20Equivalence%20of%20the%20PRESS%20Shortcut%20vs.%20Brute-Force%20LOOCV**%3A%20We%20explicitly%20fit%20%24n%20%3D%2035%24%20separate%20linear%20regression%20models%2C%20dropping%20one%20point%20at%20a%20time%2C%20and%20verify%20that%20the%20brute-force%20LOOCV%20prediction%20error%20matches%20the%20Sherman-Morrison%20shortcut%20formula%20%24%5Cfrac%7Be_i%7D%7B1%20-%20h_%7Bii%7D%7D%24%20to%20machine%20precision%20(%2410%5E%7B-12%7D%24).%0A%20%20%20%202.%20**Comprehensive%20Model%20Order%20Selection%20Table**%3A%20Evaluating%20models%20of%20degree%201%20through%206%20across%20SSE%2C%20PRESS%2C%20%24R%5E2%2C%20%5Cbar%7BR%7D%5E2%24%2C%20and%20%24R%5E2_%7B%5Ctext%7Bpred%7D%7D%24%2C%20proving%20that%20Predictive%20%24R%5E2%24%20prevents%20model%20overparameterization.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(LinearRegression%2C%20np%2C%20pd%2C%20x_raw%2C%20y_raw)%3A%0A%20%20%20%20%23%20Example%201%3A%20Numerical%20Verification%20of%20the%20PRESS%20Shortcut%20vs.%20Brute-Force%20LOOCV%0A%20%20%20%20_n%20%3D%20len(x_raw)%0A%20%20%20%20_X%20%3D%20np.column_stack(%5Bnp.ones(_n)%2C%20x_raw%2C%20x_raw**2%2C%20x_raw**3%5D)%0A%20%20%20%20_p%20%3D%20_X.shape%5B1%5D%0A%0A%20%20%20%20%23%20Standard%20full%20fit%0A%20%20%20%20_beta%20%3D%20np.linalg.solve(_X.T%20%40%20_X%2C%20_X.T%20%40%20y_raw)%0A%20%20%20%20_y_hat%20%3D%20_X%20%40%20_beta%0A%20%20%20%20_residuals%20%3D%20y_raw%20-%20_y_hat%0A%0A%20%20%20%20%23%20Hat%20matrix%20diagonal%0A%20%20%20%20_H%20%3D%20_X%20%40%20np.linalg.solve(_X.T%20%40%20_X%2C%20_X.T)%0A%20%20%20%20_h_diag%20%3D%20np.diag(_H)%0A%0A%20%20%20%20%23%201.%20Shortcut%20PRESS%20residuals%0A%20%20%20%20_press_shortcut%20%3D%20_residuals%20%2F%20(1.0%20-%20_h_diag)%0A%20%20%20%20_press_stat_shortcut%20%3D%20np.sum(_press_shortcut**2)%0A%0A%20%20%20%20%23%202.%20Brute-force%20n-fold%20Leave-One-Out%20Cross-Validation%0A%20%20%20%20_brute_force_errors%20%3D%20np.zeros(_n)%0A%20%20%20%20for%20_i%20in%20range(_n)%3A%0A%20%20%20%20%20%20%20%20_mask%20%3D%20np.ones(_n%2C%20dtype%3Dbool)%0A%20%20%20%20%20%20%20%20_mask%5B_i%5D%20%3D%20False%0A%20%20%20%20%20%20%20%20_X_train%20%3D%20_X%5B_mask%5D%0A%20%20%20%20%20%20%20%20_y_train%20%3D%20y_raw%5B_mask%5D%0A%20%20%20%20%20%20%20%20_X_test%20%3D%20_X%5B_i%20%3A%20_i%20%2B%201%5D%0A%0A%20%20%20%20%20%20%20%20_model%20%3D%20LinearRegression(fit_intercept%3DFalse).fit(_X_train%2C%20_y_train)%0A%20%20%20%20%20%20%20%20_pred_loo%20%3D%20_model.predict(_X_test)%5B0%5D%0A%20%20%20%20%20%20%20%20_brute_force_errors%5B_i%5D%20%3D%20y_raw%5B_i%5D%20-%20_pred_loo%0A%0A%20%20%20%20_press_stat_bruteforce%20%3D%20np.sum(_brute_force_errors**2)%0A%20%20%20%20_max_discrepancy%20%3D%20np.max(np.abs(_press_shortcut%20-%20_brute_force_errors))%0A%0A%20%20%20%20_df_verification%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%22Evaluation%20Method%22%3A%20%22PRESS%20Shortcut%3A%20e_i%20%2F%20(1%20-%20h_ii)%22%2C%20%22PRESS%20Statistic%22%3A%20f%22%7B_press_stat_shortcut%3A.6f%7D%22%2C%20%22Compute%20Time%20%2F%20Complexity%22%3A%20%22O(n%20p%5E2)%20%5BSingle%20OLS%20Fit%5D%22%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%22Evaluation%20Method%22%3A%20%22Brute-Force%20LOOCV%3A%20n%20re-fits%22%2C%20%22PRESS%20Statistic%22%3A%20f%22%7B_press_stat_bruteforce%3A.6f%7D%22%2C%20%22Compute%20Time%20%2F%20Complexity%22%3A%20%22O(n%5E2%20p%5E2)%20%5B35%20separate%20models%5D%22%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%22Evaluation%20Method%22%3A%20%22Max%20Absolute%20Error%20Discrepancy%22%2C%20%22PRESS%20Statistic%22%3A%20f%22%7B_max_discrepancy%3A.2e%7D%22%2C%20%22Compute%20Time%20%2F%20Complexity%22%3A%20%22Exact%20to%20Machine%20Precision%22%7D%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(np%2C%20pd%2C%20x_raw%2C%20y_raw)%3A%0A%20%20%20%20%23%20Example%202%3A%20Model%20Order%20Selection%20Diagnostic%20Table%20(Degrees%201%20to%206)%0A%20%20%20%20_n%20%3D%20len(x_raw)%0A%20%20%20%20_ss_tot%20%3D%20np.sum((y_raw%20-%20np.mean(y_raw))%20**%202)%0A%0A%20%20%20%20_eval_rows%20%3D%20%5B%5D%0A%20%20%20%20for%20_deg%20in%20range(1%2C%207)%3A%0A%20%20%20%20%20%20%20%20_p%20%3D%20_deg%20%2B%201%0A%20%20%20%20%20%20%20%20_X%20%3D%20np.vander(x_raw%2C%20_p)%0A%20%20%20%20%20%20%20%20_beta%20%3D%20np.linalg.solve(_X.T%20%40%20_X%2C%20_X.T%20%40%20y_raw)%0A%20%20%20%20%20%20%20%20_y_pred%20%3D%20_X%20%40%20_beta%0A%20%20%20%20%20%20%20%20_e%20%3D%20y_raw%20-%20_y_pred%0A%0A%20%20%20%20%20%20%20%20_H%20%3D%20_X%20%40%20np.linalg.solve(_X.T%20%40%20_X%2C%20_X.T)%0A%20%20%20%20%20%20%20%20_h%20%3D%20np.diag(_H)%0A%0A%20%20%20%20%20%20%20%20_sse%20%3D%20np.sum(_e**2)%0A%20%20%20%20%20%20%20%20_press%20%3D%20np.sum((_e%20%2F%20(1.0%20-%20_h))%20**%202)%0A%0A%20%20%20%20%20%20%20%20_r2%20%3D%201.0%20-%20(_sse%20%2F%20_ss_tot)%0A%20%20%20%20%20%20%20%20_adj_r2%20%3D%201.0%20-%20(1.0%20-%20_r2)%20*%20(_n%20-%201)%20%2F%20(_n%20-%20_p)%0A%20%20%20%20%20%20%20%20_pred_r2%20%3D%201.0%20-%20(_press%20%2F%20_ss_tot)%0A%0A%20%20%20%20%20%20%20%20_status%20%3D%20%22Optimal%20Model%22%20if%20_deg%20%3D%3D%203%20else%20(%22Underfitting%22%20if%20_deg%20%3C%203%20else%20%22Overfitting%20(Generalization%20Drops)%22)%0A%0A%20%20%20%20%20%20%20%20_eval_rows.append(%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Polynomial%20Order%22%3A%20f%22Degree%20%7B_deg%7D%20(p%20%3D%20%7B_p%7D)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22SSE%20(In-sample)%22%3A%20f%22%7B_sse%3A.2f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22PRESS%20(Out-of-sample)%22%3A%20f%22%7B_press%3A.2f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Raw%20R%5E2%22%3A%20f%22%7B_r2%3A.4f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Adjusted%20R%5E2%22%3A%20f%22%7B_adj_r2%3A.4f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Predictive%20R%5E2%22%3A%20f%22%7B_pred_r2%3A.4f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Generalization%20Verdict%22%3A%20_status%2C%0A%20%20%20%20%20%20%20%20%7D)%0A%0A%20%20%20%20_df_order_selection%20%3D%20pd.DataFrame(_eval_rows)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
0bd078606dad1bdf69c836564f805d33