import%20marimo%0A%0A__generated_with%20%3D%20%220.24.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20math%0A%20%20%20%20import%20marimo%20as%20mo%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20pandas%20as%20pd%0A%20%20%20%20import%20plotly.graph_objects%20as%20go%0A%20%20%20%20import%20torch%0A%20%20%20%20import%20torch.nn.functional%20as%20F%0A%20%20%20%20from%20plotly.subplots%20import%20make_subplots%0A%20%20%20%20from%20scipy.special%20import%20erf%0A%0A%20%20%20%20return%20F%2C%20erf%2C%20go%2C%20make_subplots%2C%20mo%2C%20np%2C%20pd%2C%20torch%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%5B%E2%86%90%2046%20Model%20Counterfactuals%5D(46_model_counterfactuals.py)%20%7C%20%5BIndex%5D(..%2Findex.html)%20%7C%20%5B48%20Temperature%20Scaled%20Softmax%20%E2%86%92%5D(48_temperature_scaled_softmax.py)%0A%0A%20%20%20%20%23%20Gaussian%20Error%20Linear%20Unit%20(GELU)%3A%20Stochastic%20Regularization%2C%20Error%20Functions%2C%20and%20Transformer%20Activations%0A%0A%20%20%20%20%23%23%20%5Ba%5D%20Why%20do%20you%20need%20to%20know%20these%20concepts%3F%0A%0A%20%20%20%20For%20over%20a%20decade%2C%20the%20Rectified%20Linear%20Unit%20(%24%5Ctext%7BReLU%7D(x)%20%3D%20%5Cmax(0%2C%20x)%24)%20served%20as%20the%20standard%20non-linear%20activation%20across%20computer%20vision%20and%20neural%20network%20architectures.%20However%2C%20deep%20neural%20networks%20and%20attention-based%20Transformer%20models%20suffer%20from%20two%20structural%20shortcomings%20inherent%20to%20ReLU%3A%0A%0A%20%20%20%20%23%23%23%23%201.%20The%20Dying%20ReLU%20Pathology%0A%20%20%20%20Because%20the%20derivative%20of%20ReLU%20is%20identically%20zero%20for%20all%20negative%20pre-activations%20(%24%5Cfrac%7Bd%7D%7Bdx%7D%5Ctext%7BReLU%7D%20%3D%200%20%5C%20%5Cforall%20x%20%3C%200%24)%2C%20any%20gradient%20update%20that%20pushes%20a%20neuron's%20weights%20into%20the%20negative%20domain%20leaves%20that%20neuron%20permanently%20deactivated.%20It%20emits%20zero%20output%20and%20zero%20gradient%20for%20all%20subsequent%20training%20examples%2C%20shrinking%20the%20effective%20representational%20capacity%20of%20the%20model.%0A%0A%20%20%20%20%23%23%23%23%202.%20The%20Non-Differentiability%20Kink%20at%20%24x%20%3D%200%24%0A%20%20%20%20ReLU%20has%20a%20non-differentiable%20sharp%20corner%20at%20%24x%20%3D%200%24.%20In%20deep%20Transformer%20architectures%20containing%20dozens%20or%20hundreds%20of%20stacked%20multi-head%20self-attention%20and%20feed-forward%20layers%2C%20these%20non-smooth%20points%20create%20optimization%20friction%20and%20gradient%20instability.%0A%0A%20%20%20%20%23%23%23%23%20The%20GELU%20Innovation%20in%20Transformers%0A%20%20%20%20Introduced%20by%20Dan%20Hendrycks%20and%20Kevin%20Gimpel%20in%202016%2C%20the%20**Gaussian%20Error%20Linear%20Unit%20(GELU)**%20bridges%20deterministic%20non-linear%20activation%20with%20stochastic%20regularization%20(Dropout)%3A%0A%20%20%20%20-%20Instead%20of%20deterministically%20zeroing%20negative%20inputs%20based%20on%20an%20arbitrary%20step%20cutoff%20(%24x%20%3E%200%24)%2C%20GELU%20weights%20an%20input%20by%20the%20probability%20that%20a%20standard%20normal%20variable%20is%20less%20than%20%24x%24.%0A%20%20%20%20-%20Large%20positive%20inputs%20are%20preserved%20almost%20linearly%20(%24x%20%5CPhi(x)%20%5Capprox%20x%24).%0A%20%20%20%20-%20Large%20negative%20inputs%20are%20suppressed%20smoothly%20toward%20zero%20(%24x%20%5CPhi(x)%20%5Capprox%200%24).%0A%20%20%20%20-%20Moderately%20negative%20inputs%20retain%20a%20small%2C%20smooth%20negative%20trough%20(%24%5Cmin%20%5Capprox%20-0.17%24)%2C%20allowing%20gradient%20flow%20even%20when%20pre-activations%20dip%20below%20zero.%0A%0A%20%20%20%20Because%20of%20its%20smooth%20curvature%20(%24%5Cmathcal%7BC%7D%5E%5Cinfty%24%20differentiability)%20and%20superior%20empirical%20performance%2C%20GELU%20was%20selected%20as%20the%20default%20feed-forward%20activation%20function%20for%20**BERT%2C%20RoBERTa%2C%20GPT-2%2C%20GPT-3%2C%20GPT-4%2C%20and%20Vision%20Transformers%20(ViT)**.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20%5Bb%5D%20Mathematical%20Foundations%20and%20Analytical%20Derivations%0A%0A%20%20%20%20%23%23%23%201.%20Probabilistic%20Formulation%0A%0A%20%20%20%20Let%20%24x%20%5Cin%20%5Cmathbb%7BR%7D%24%20denote%20a%20scalar%20neuron%20pre-activation.%20Suppose%20we%20multiply%20%24x%24%20by%20a%20stochastic%20Bernoulli%20gate%20%24m%20%5Csim%20%5Coperatorname%7BBernoulli%7D(%5Cpi(x))%24%2C%20where%20the%20activation%20probability%20is%20dictated%20by%20the%20Cumulative%20Distribution%20Function%20(CDF)%20of%20a%20standard%20normal%20distribution%3A%0A%0A%20%20%20%20%24%24%5Cpi(x)%20%3D%20P(Z%20%5Cle%20x)%20%3D%20%5CPhi(x)%2C%20%5Cquad%20%5Ctext%7Bwhere%20%7D%20Z%20%5Csim%20%5Cmathcal%7BN%7D(0%2C%201)%24%24%0A%0A%20%20%20%20The%20deterministic%20activation%20function%20is%20defined%20as%20the%20mathematical%20expectation%20of%20this%20stochastic%20gating%20process%3A%0A%0A%20%20%20%20%24%24%5Coperatorname%7BGELU%7D(x)%20%3D%20%5Cmathbb%7BE%7D%5Bm%20%5Ccdot%20x%5D%20%3D%20x%20%5Ccdot%20P(Z%20%5Cle%20x)%20%3D%20x%20%5CPhi(x)%24%24%0A%0A%20%20%20%20The%20cumulative%20distribution%20function%20%24%5CPhi(x)%24%20of%20the%20standard%20normal%20distribution%20is%3A%0A%0A%20%20%20%20%24%24%5CPhi(x)%20%3D%20%5Cfrac%7B1%7D%7B%5Csqrt%7B2%5Cpi%7D%7D%20%5Cint_%7B-%5Cinfty%7D%5Ex%20e%5E%7B-%5Cfrac%7Bt%5E2%7D%7B2%7D%7D%20dt%20%3D%20%5Cfrac%7B1%7D%7B2%7D%20%5Cleft%5B%201%20%2B%20%5Coperatorname%7Berf%7D%5Cleft(%20%5Cfrac%7Bx%7D%7B%5Csqrt%7B2%7D%7D%20%5Cright)%20%5Cright%5D%24%24%0A%0A%20%20%20%20where%20%24%5Coperatorname%7Berf%7D(z)%24%20is%20the%20standard%20Gauss%20error%20function%3A%0A%0A%20%20%20%20%24%24%5Coperatorname%7Berf%7D(z)%20%3D%20%5Cfrac%7B2%7D%7B%5Csqrt%7B%5Cpi%7D%7D%20%5Cint_0%5Ez%20e%5E%7B-t%5E2%7D%20dt%24%24%0A%0A%20%20%20%20Substituting%20the%20error%20function%20into%20the%20definition%20yields%20the%20**exact%20formulation**%3A%0A%0A%20%20%20%20%24%24%5Coperatorname%7BGELU%7D(x)%20%3D%20%5Cfrac%7B1%7D%7B2%7D%20x%20%5Cleft%5B%201%20%2B%20%5Coperatorname%7Berf%7D%5Cleft(%20%5Cfrac%7Bx%7D%7B%5Csqrt%7B2%7D%7D%20%5Cright)%20%5Cright%5D%24%24%0A%0A%20%20%20%20%23%23%23%202.%20Fast%20Approximations%0A%0A%20%20%20%20Because%20evaluating%20the%20continuous%20error%20function%20%24%5Coperatorname%7Berf%7D(x)%24%20requires%20numerical%20integration%20or%20polynomial%20series%20expansions%2C%20deep%20learning%20frameworks%20provide%20optimized%20approximations%3A%0A%0A%20%20%20%20%23%23%23%23%20The%20Tanh%20Approximation%20(Hendrycks%20%26%20Gimpel%2C%202016)%0A%20%20%20%20Used%20in%20the%20original%20OpenAI%20GPT%20and%20Google%20BERT%20codebases%3A%0A%0A%20%20%20%20%24%24%5Coperatorname%7BGELU%7D_%7B%5Ctext%7Btanh%7D%7D(x)%20%3D%20%5Cfrac%7B1%7D%7B2%7D%20x%20%5Cleft%5B%201%20%2B%20%5Ctanh%5Cleft(%20%5Csqrt%7B%5Cfrac%7B2%7D%7B%5Cpi%7D%7D%20%5Cleft(%20x%20%2B%200.044715%20x%5E3%20%5Cright)%20%5Cright)%20%5Cright%5D%24%24%0A%0A%20%20%20%20The%20maximum%20absolute%20error%20between%20%24%5Coperatorname%7BGELU%7D_%7B%5Ctext%7Bexact%7D%7D(x)%24%20and%20%24%5Coperatorname%7BGELU%7D_%7B%5Ctext%7Btanh%7D%7D(x)%24%20is%20less%20than%20%241.4%20%5Ctimes%2010%5E%7B-4%7D%24%20across%20all%20%24x%20%5Cin%20%5Cmathbb%7BR%7D%24.%0A%0A%20%20%20%20%23%23%23%23%20The%20Sigmoid%20Approximation%0A%20%20%20%20%24%24%5Coperatorname%7BGELU%7D_%7B%5Ctext%7Bsigmoid%7D%7D(x)%20%3D%20x%20%5Ccdot%20%5Csigma(1.702%20x)%20%3D%20%5Cfrac%7Bx%7D%7B1%20%2B%20e%5E%7B-1.702%20x%7D%7D%24%24%0A%0A%20%20%20%20%23%23%23%203.%20First%20and%20Second%20Derivatives%0A%0A%20%20%20%20Applying%20the%20product%20rule%20of%20calculus%20to%20%24%5Coperatorname%7BGELU%7D(x)%20%3D%20x%20%5CPhi(x)%24%3A%0A%0A%20%20%20%20%24%24%5Cfrac%7Bd%7D%7Bdx%7D%20%5Coperatorname%7BGELU%7D(x)%20%3D%20%5Cfrac%7Bd%7D%7Bdx%7D%5Bx%5D%20%5Ccdot%20%5CPhi(x)%20%2B%20x%20%5Ccdot%20%5Cfrac%7Bd%7D%7Bdx%7D%5B%5CPhi(x)%5D%20%3D%20%5CPhi(x)%20%2B%20x%20%5Cphi(x)%24%24%0A%0A%20%20%20%20where%20%24%5Cphi(x)%20%3D%20%5CPhi'(x)%20%3D%20%5Cfrac%7B1%7D%7B%5Csqrt%7B2%5Cpi%7D%7D%20e%5E%7B-x%5E2%2F2%7D%24%20is%20the%20standard%20normal%20Probability%20Density%20Function%20(PDF).%0A%0A%20%20%20%20Substituting%20the%20error%20function%3A%0A%0A%20%20%20%20%24%24%5Cfrac%7Bd%7D%7Bdx%7D%20%5Coperatorname%7BGELU%7D(x)%20%3D%20%5Cfrac%7B1%7D%7B2%7D%5Cleft%5B%201%20%2B%20%5Coperatorname%7Berf%7D%5Cleft(%20%5Cfrac%7Bx%7D%7B%5Csqrt%7B2%7D%7D%20%5Cright)%20%5Cright%5D%20%2B%20%5Cfrac%7Bx%7D%7B%5Csqrt%7B2%5Cpi%7D%7D%20e%5E%7B-%5Cfrac%7Bx%5E2%7D%7B2%7D%7D%24%24%0A%0A%20%20%20%20%23%23%23%23%20Asymptotic%20Behavior%0A%20%20%20%20-%20**As%20%24x%20%5Cto%20%2B%5Cinfty%24**%3A%20%24%5CPhi(x)%20%5Cto%201%24%20and%20%24x%20%5Cphi(x)%20%5Cto%200%24%2C%20so%20%24%5Cfrac%7Bd%7D%7Bdx%7D%20%5Coperatorname%7BGELU%7D(x)%20%5Cto%201%24.%20Large%20activations%20pass%20gradients%20through%20with%20unit%20gain%2C%20preventing%20vanishing%20gradients.%0A%20%20%20%20-%20**As%20%24x%20%5Cto%20-%5Cinfty%24**%3A%20%24%5CPhi(x)%20%5Cto%200%24%20and%20%24x%20%5Cphi(x)%20%5Cto%200%24%2C%20so%20%24%5Cfrac%7Bd%7D%7Bdx%7D%20%5Coperatorname%7BGELU%7D(x)%20%5Cto%200%24.%0A%20%20%20%20-%20**At%20%24x%20%3D%200%24**%3A%20%24%5CPhi(0)%20%3D%200.5%24%20and%20%240%20%5Ccdot%20%5Cphi(0)%20%3D%200%24%2C%20so%20the%20derivative%20is%20exactly%20%24%5Cfrac%7Bd%7D%7Bdx%7D%20%5Coperatorname%7BGELU%7D(0)%20%3D%200.5%24.%0A%20%20%20%20-%20**Minimum%20Stationary%20Point**%3A%20The%20function%20reaches%20its%20local%20minimum%20at%20%24x%5E*%20%5Capprox%20-0.7518%24%2C%20where%20%24%5Coperatorname%7BGELU%7D(x%5E*)%20%5Capprox%20-0.1699%24.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(erf%2C%20np)%3A%0A%20%20%20%20%23%20Coordinate%20grid%20across%20typical%20activation%20pre-activation%20range%0A%20%20%20%20x_grid%20%3D%20np.linspace(-4.0%2C%204.0%2C%20500)%0A%0A%20%20%20%20%23%201.%20Exact%20GELU%3A%200.5%20*%20x%20*%20(1%20%2B%20erf(x%20%2F%20sqrt(2)))%0A%20%20%20%20gelu_exact%20%3D%200.5%20*%20x_grid%20*%20(1.0%20%2B%20erf(x_grid%20%2F%20np.sqrt(2.0)))%0A%0A%20%20%20%20%23%202.%20Tanh%20Approximation%0A%20%20%20%20gelu_tanh%20%3D%200.5%20*%20x_grid%20*%20(%0A%20%20%20%20%20%20%20%201.0%20%2B%20np.tanh(np.sqrt(2.0%20%2F%20np.pi)%20*%20(x_grid%20%2B%200.044715%20*%20(x_grid**3)))%0A%20%20%20%20)%0A%0A%20%20%20%20%23%203.%20Standard%20ReLU%3A%20max(0%2C%20x)%0A%20%20%20%20relu_curve%20%3D%20np.maximum(0.0%2C%20x_grid)%0A%0A%20%20%20%20%23%204.%20Leaky%20ReLU%3A%20max(0.01%20*%20x%2C%20x)%0A%20%20%20%20leaky_relu_curve%20%3D%20np.where(x_grid%20%3E%200%2C%20x_grid%2C%200.08%20*%20x_grid)%0A%0A%20%20%20%20%23%205.%20SiLU%20(Swish)%3A%20x%20*%20sigmoid(x)%0A%20%20%20%20silu_curve%20%3D%20x_grid%20%2F%20(1.0%20%2B%20np.exp(-x_grid))%0A%0A%20%20%20%20%23%20Derivative%20computations%0A%20%20%20%20%23%20Exact%20GELU%20derivative%3A%20Phi(x)%20%2B%20x%20*%20phi(x)%0A%20%20%20%20norm_cdf%20%3D%200.5%20*%20(1.0%20%2B%20erf(x_grid%20%2F%20np.sqrt(2.0)))%0A%20%20%20%20norm_pdf%20%3D%20(1.0%20%2F%20np.sqrt(2.0%20*%20np.pi))%20*%20np.exp(-0.5%20*%20(x_grid**2))%0A%20%20%20%20gelu_derivative%20%3D%20norm_cdf%20%2B%20x_grid%20*%20norm_pdf%0A%0A%20%20%20%20%23%20ReLU%20derivative%3A%20Heaviside%20step%0A%20%20%20%20relu_derivative%20%3D%20np.where(x_grid%20%3E%200%2C%201.0%2C%200.0)%0A%0A%20%20%20%20%23%20SiLU%20derivative%3A%20sigma(x)%20%2B%20x%20*%20sigma(x)%20*%20(1%20-%20sigma(x))%0A%20%20%20%20sig_x%20%3D%201.0%20%2F%20(1.0%20%2B%20np.exp(-x_grid))%0A%20%20%20%20silu_derivative%20%3D%20sig_x%20%2B%20x_grid%20*%20sig_x%20*%20(1.0%20-%20sig_x)%0A%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20gelu_derivative%2C%0A%20%20%20%20%20%20%20%20gelu_exact%2C%0A%20%20%20%20%20%20%20%20gelu_tanh%2C%0A%20%20%20%20%20%20%20%20leaky_relu_curve%2C%0A%20%20%20%20%20%20%20%20relu_curve%2C%0A%20%20%20%20%20%20%20%20relu_derivative%2C%0A%20%20%20%20%20%20%20%20silu_curve%2C%0A%20%20%20%20%20%20%20%20silu_derivative%2C%0A%20%20%20%20%20%20%20%20x_grid%2C%0A%20%20%20%20)%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20gelu_derivative%2C%0A%20%20%20%20gelu_exact%2C%0A%20%20%20%20go%2C%0A%20%20%20%20leaky_relu_curve%2C%0A%20%20%20%20make_subplots%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20np%2C%0A%20%20%20%20relu_curve%2C%0A%20%20%20%20relu_derivative%2C%0A%20%20%20%20silu_curve%2C%0A%20%20%20%20silu_derivative%2C%0A%20%20%20%20x_grid%2C%0A)%3A%0A%20%20%20%20fig%20%3D%20make_subplots(%0A%20%20%20%20%20%20%20%20rows%3D1%2C%0A%20%20%20%20%20%20%20%20cols%3D2%2C%0A%20%20%20%20%20%20%20%20subplot_titles%3D%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%3Cb%3EActivation%20Function%20Geometries%3A%20GELU%20vs%20Classical%20Defaults%3C%2Fb%3E%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%3Cb%3EActivation%20Derivatives%20(Gradient%20Flow%20Profiles%20dy%2Fdx)%3C%2Fb%3E%22%2C%0A%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%20%20%20%20horizontal_spacing%3D0.12%2C%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Left%3A%20Activation%20curves%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dx_grid%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dgelu_exact%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%232563EB%22%2C%20width%3D2.5)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22GELU%20(Exact)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D1%2C%0A%20%20%20%20)%0A%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dx_grid%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Drelu_curve%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%23DC2626%22%2C%20width%3D2%2C%20dash%3D%22dash%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22ReLU%3A%20max(0%2C%20x)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D1%2C%0A%20%20%20%20)%0A%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dx_grid%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dsilu_curve%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%2310B981%22%2C%20width%3D2%2C%20dash%3D%22dot%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22SiLU%20(Swish)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D1%2C%0A%20%20%20%20)%0A%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dx_grid%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dleaky_relu_curve%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%23F59E0B%22%2C%20width%3D1.5%2C%20dash%3D%22dashdot%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Leaky%20ReLU%20(alpha%3D0.08)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D1%2C%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Mark%20GELU%20global%20minimum%0A%20%20%20%20min_idx%20%3D%20np.argmin(gelu_exact)%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3D%5Bx_grid%5Bmin_idx%5D%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3D%5Bgelu_exact%5Bmin_idx%5D%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22markers%2Btext%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20marker%3Ddict(color%3D%22%231E3A8A%22%2C%20size%3D8)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20text%3D%5Bf%22Min%20(%7Bx_grid%5Bmin_idx%5D%3A.2f%7D%2C%20%7Bgelu_exact%5Bmin_idx%5D%3A.2f%7D)%22%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20textposition%3D%22bottom%20center%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20showlegend%3DFalse%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D1%2C%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Right%3A%20Derivatives%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dx_grid%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dgelu_derivative%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%232563EB%22%2C%20width%3D2.5)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22GELU%20Derivative%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dx_grid%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Drelu_derivative%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%23DC2626%22%2C%20width%3D2%2C%20dash%3D%22dash%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22ReLU%20Derivative%20(Step)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dx_grid%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dsilu_derivative%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%2310B981%22%2C%20width%3D2%2C%20dash%3D%22dot%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22SiLU%20Derivative%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%0A%20%20%20%20fig.update_xaxes(title_text%3D%22Input%20Pre-Activation%20x%22%2C%20row%3D1%2C%20col%3D1)%0A%20%20%20%20fig.update_yaxes(title_text%3D%22Activated%20Output%20f(x)%22%2C%20row%3D1%2C%20col%3D1)%0A%20%20%20%20fig.update_xaxes(title_text%3D%22Input%20Pre-Activation%20x%22%2C%20row%3D1%2C%20col%3D2)%0A%20%20%20%20fig.update_yaxes(title_text%3D%22Derivative%20df%2Fdx%22%2C%20range%3D%5B-0.2%2C%201.2%5D%2C%20row%3D1%2C%20col%3D2)%0A%0A%20%20%20%20fig.update_layout(%0A%20%20%20%20%20%20%20%20template%3D%22plotly_white%22%2C%0A%20%20%20%20%20%20%20%20height%3D500%2C%0A%20%20%20%20%20%20%20%20margin%3Ddict(l%3D40%2C%20r%3D40%2C%20t%3D70%2C%20b%3D50)%2C%0A%20%20%20%20%20%20%20%20legend%3Ddict(orientation%3D%22h%22%2C%20yanchor%3D%22bottom%22%2C%20y%3D-0.28%2C%20xanchor%3D%22center%22%2C%20x%3D0.5)%2C%0A%20%20%20%20)%0A%0A%20%20%20%20viz%20%3D%20mo.ui.plotly(fig)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(F%2C%20erf%2C%20gelu_exact%2C%20gelu_tanh%2C%20mo%2C%20np%2C%20pd%2C%20torch%2C%20x_grid)%3A%0A%20%20%20%20%23%20Example%201%3A%20Precision%20Benchmarking%3A%20Exact%20Formula%20vs%20Tanh%20vs%20PyTorch%20Built-in%0A%20%20%20%20torch_x%20%3D%20torch.tensor(x_grid%2C%20dtype%3Dtorch.float64%2C%20requires_grad%3DTrue)%0A%20%20%20%20torch_gelu_exact%20%3D%20F.gelu(torch_x%2C%20approximate%3D%22none%22).detach().numpy()%0A%20%20%20%20torch_gelu_tanh%20%3D%20F.gelu(torch_x%2C%20approximate%3D%22tanh%22).detach().numpy()%0A%0A%20%20%20%20max_err_exact_vs_torch%20%3D%20float(np.max(np.abs(gelu_exact%20-%20torch_gelu_exact)))%0A%20%20%20%20max_err_tanh_vs_torch%20%3D%20float(np.max(np.abs(gelu_tanh%20-%20torch_gelu_tanh)))%0A%20%20%20%20max_err_tanh_vs_exact%20%3D%20float(np.max(np.abs(gelu_tanh%20-%20gelu_exact)))%0A%0A%20%20%20%20df_precision%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Comparison%22%3A%20%22From-Scratch%20Exact%20vs%20PyTorch%20F.gelu('none')%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Formula%22%3A%20%220.5%20*%20x%20*%20(1%20%2B%20erf(x%20%2F%20sqrt(2)))%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Max_Absolute_Error%22%3A%20f%22%7Bmax_err_exact_vs_torch%3A.2e%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Implementation_Status%22%3A%20%22Identical%20down%20to%20machine%20epsilon%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Comparison%22%3A%20%22From-Scratch%20Tanh%20vs%20PyTorch%20F.gelu('tanh')%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Formula%22%3A%20%220.5%20*%20x%20*%20(1%20%2B%20tanh(sqrt(2%2Fpi)%20*%20(x%20%2B%200.044715%20x%5E3)))%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Max_Absolute_Error%22%3A%20f%22%7Bmax_err_tanh_vs_torch%3A.2e%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Implementation_Status%22%3A%20%22Identical%20down%20to%20machine%20epsilon%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Comparison%22%3A%20%22Tanh%20Approximation%20vs%20Exact%20GELU%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Formula%22%3A%20%22Discrepancy%20of%20Hendrycks%20%26%20Gimpel%20fast%20form%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Max_Absolute_Error%22%3A%20f%22%7Bmax_err_tanh_vs_exact%3A.2e%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Implementation_Status%22%3A%20%22Within%20theoretical%20bound%20(%3C%201.4e-4)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Example%202%3A%20Analytical%20Derivative%20vs%20PyTorch%20Autograd%20Backward%20Pass%0A%20%20%20%20torch_x_grad%20%3D%20torch.tensor(x_grid%2C%20dtype%3Dtorch.float64%2C%20requires_grad%3DTrue)%0A%20%20%20%20y_torch%20%3D%20F.gelu(torch_x_grad%2C%20approximate%3D%22none%22)%0A%20%20%20%20y_torch.backward(torch.ones_like(torch_x_grad))%0A%20%20%20%20autograd_deriv%20%3D%20torch_x_grad.grad.detach().numpy()%0A%0A%20%20%20%20df_derivative_verif%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Test_Point_x%22%3A%20np.array(%5B-3.0%2C%20-1.5%2C%20-0.75%2C%200.0%2C%200.75%2C%201.5%2C%203.0%5D)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Analytical_Derivative%22%3A%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%200.5%20*%20(1.0%20%2B%20erf(np.array(%5B-3.0%2C%20-1.5%2C%20-0.75%2C%200.0%2C%200.75%2C%201.5%2C%203.0%5D)%20%2F%20np.sqrt(2.0)))%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%2B%20np.array(%5B-3.0%2C%20-1.5%2C%20-0.75%2C%200.0%2C%200.75%2C%201.5%2C%203.0%5D)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20*%20(1.0%20%2F%20np.sqrt(2.0%20*%20np.pi))%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20*%20np.exp(-0.5%20*%20np.array(%5B-3.0%2C%20-1.5%2C%20-0.75%2C%200.0%2C%200.75%2C%201.5%2C%203.0%5D)%20**%202)%0A%20%20%20%20%20%20%20%20%20%20%20%20).round(5)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22PyTorch_Autograd_dy_dx%22%3A%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20round(autograd_deriv%5Bnp.abs(x_grid%20-%20pt).argmin()%5D%2C%205)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20pt%20in%20%5B-3.0%2C%20-1.5%2C%20-0.75%2C%200.0%2C%200.75%2C%201.5%2C%203.0%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22Status%22%3A%20%22Exact%20Numerical%20Match%22%2C%0A%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Example%203%3A%20Activation%20Functions%20Architectural%20Comparison%20Table%0A%20%20%20%20df_archetypes%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Activation%22%3A%20%22GELU%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Mathematical_Form%22%3A%20%22x%20*%20Phi(x)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Differentiability%22%3A%20%22Smooth%20(C_inf)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Dying_Neuron_Immunity%22%3A%20%22High%20(Grad%20flow%20in%20negative%20well)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Default_Usage%22%3A%20%22BERT%2C%20GPT-2%2F3%2F4%2C%20RoBERTa%2C%20ViT%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Activation%22%3A%20%22ReLU%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Mathematical_Form%22%3A%20%22max(0%2C%20x)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Differentiability%22%3A%20%22Non-differentiable%20at%20x%3D0%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Dying_Neuron_Immunity%22%3A%20%22Zero%20(Permanent%20death%20if%20x%20%3C%200)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Default_Usage%22%3A%20%22ResNets%2C%20Early%20CNNs%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Activation%22%3A%20%22SiLU%20(Swish)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Mathematical_Form%22%3A%20%22x%20*%20sigma(x)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Differentiability%22%3A%20%22Smooth%20(C_inf)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Dying_Neuron_Immunity%22%3A%20%22High%20(Smooth%20negative%20well)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Default_Usage%22%3A%20%22LLaMA%2C%20Mistral%2C%20EfficientNet%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Activation%22%3A%20%22Leaky%20ReLU%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Mathematical_Form%22%3A%20%22max(alpha%20*%20x%2C%20x)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Differentiability%22%3A%20%22Non-differentiable%20at%20x%3D0%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Dying_Neuron_Immunity%22%3A%20%22Moderate%20(Fixed%20alpha%20slope)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Default_Usage%22%3A%20%22GAN%20Discriminators%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%0A%20%20%20%20table_precision%20%3D%20mo.ui.table(df_precision)%0A%20%20%20%20table_deriv%20%3D%20mo.ui.table(df_derivative_verif)%0A%20%20%20%20table_archetypes%20%3D%20mo.ui.table(df_archetypes)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
d54153c008bfdd3268cad0d613d4aaa1