import%20marimo%0A%0A__generated_with%20%3D%20%220.24.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20pandas%20as%20pd%0A%20%20%20%20import%20plotly.graph_objects%20as%20go%0A%20%20%20%20from%20plotly.subplots%20import%20make_subplots%0A%0A%20%20%20%20return%20go%2C%20make_subplots%2C%20mo%2C%20np%2C%20pd%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%5B%E2%86%90%2049%20Focal%20Loss%5D(49_focal_loss_balanced.py)%20%7C%20%5BIndex%5D(..%2Findex.html)%20%7C%20%5B51%20Causal%20Attention%20%E2%86%92%5D(51_causal_attention.py)%0A%0A%20%20%20%20%23%2050.%20Scaled%20Dot-Product%20Attention%3A%20The%20Mathematical%20Engine%20of%20Transformer%20Architectures%0A%0A%20%20%20%20%23%23%23%20Executive%20Summary%0A%0A%20%20%20%20The%20**Scaled%20Dot-Product%20Attention**%20mechanism%20(Vaswani%20et%20al.%2C%202017)%20represents%20the%20foundational%20computational%20building%20block%20of%20modern%20large%20language%20models%2C%20vision%20transformers%2C%20and%20multimodal%20foundational%20architectures.%20Departing%20from%20recurrence%20(RNNs)%20and%20convolution%20(CNNs)%2C%20attention%20operates%20as%20a%20content-based%20associative%20memory%20that%20constructs%20context-aware%20representations%20by%20computing%20pairwise%20affinity%20weights%20across%20arbitrary%20sequence%20lengths%20in%20%24%5Cmathcal%7BO%7D(1)%24%20sequential%20operations.%0A%0A%20%20%20%20At%20its%20core%2C%20the%20mechanism%20maps%20each%20token%20into%20three%20distinct%20linear%20projections%3A%20a%20**Query**%20(%24Q%24%2C%20what%20information%20is%20sought)%2C%20a%20**Key**%20(%24K%24%2C%20what%20information%20is%20indexed)%2C%20and%20a%20**Value**%20(%24V%24%2C%20the%20actual%20content%20retrieved).%20By%20scaling%20raw%20dot-product%20affinities%20by%20%241%2F%5Csqrt%7Bd_k%7D%24%2C%20it%20solves%20the%20severe%20vanishing%20gradient%20problem%20inherent%20in%20high-dimensional%20softmax%20transformations%2C%20ensuring%20stable%20optimization%20across%20modern%20deep%20networks.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20%5Bb%5D%20Mathematical%20Foundations%20and%20Scaling%20Derivations%0A%0A%20%20%20%20%23%23%23%201.%20Matrix%20Formulation%20of%20Scaled%20Dot-Product%20Attention%0A%0A%20%20%20%20Let%20%24X%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BN%20%5Ctimes%20d_%7B%5Ctext%7Bmodel%7D%7D%7D%24%20represent%20an%20input%20sequence%20of%20%24N%24%20token%20embeddings.%20Given%20learnable%20projection%20matrices%20%24W_Q%20%5Cin%20%5Cmathbb%7BR%7D%5E%7Bd_%7B%5Ctext%7Bmodel%7D%7D%20%5Ctimes%20d_k%7D%24%2C%20%24W_K%20%5Cin%20%5Cmathbb%7BR%7D%5E%7Bd_%7B%5Ctext%7Bmodel%7D%7D%20%5Ctimes%20d_k%7D%24%2C%20and%20%24W_V%20%5Cin%20%5Cmathbb%7BR%7D%5E%7Bd_%7B%5Ctext%7Bmodel%7D%7D%20%5Ctimes%20d_v%7D%24%2C%20the%20query%2C%20key%2C%20and%20value%20matrices%20are%3A%0A%0A%20%20%20%20%24%24Q%20%3D%20X%20W_Q%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BN%20%5Ctimes%20d_k%7D%2C%20%5Cquad%20K%20%3D%20X%20W_K%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BM%20%5Ctimes%20d_k%7D%2C%20%5Cquad%20V%20%3D%20X%20W_V%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BM%20%5Ctimes%20d_v%7D%24%24%0A%0A%20%20%20%20The%20scaled%20dot-product%20attention%20output%20%24Y%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BN%20%5Ctimes%20d_v%7D%24%20is%20defined%20by%20the%20compact%20matrix%20equation%3A%0A%0A%20%20%20%20%24%24%5Coperatorname%7BAttention%7D(Q%2C%20K%2C%20V)%20%3D%20%5Coperatorname%7BSoftmax%7D%5Cleft(%5Cfrac%7BQ%20K%5E%5Ctop%7D%7B%5Csqrt%7Bd_k%7D%7D%5Cright)%20V%24%24%0A%0A%20%20%20%20where%20the%20pre-softmax%20score%20matrix%20%24S%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BN%20%5Ctimes%20M%7D%24%20and%20attention%20weight%20matrix%20%24A%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BN%20%5Ctimes%20M%7D%24%20are%3A%0A%0A%20%20%20%20%24%24S%20%3D%20%5Cfrac%7BQ%20K%5E%5Ctop%7D%7B%5Csqrt%7Bd_k%7D%7D%2C%20%5Cqquad%20A_%7Bij%7D%20%3D%20%5Cfrac%7B%5Cexp(S_%7Bij%7D)%7D%7B%5Csum_%7Bk%3D1%7D%5EM%20%5Cexp(S_%7Bik%7D)%7D%2C%20%5Cqquad%20Y%20%3D%20A%20V%24%24%0A%0A%20%20%20%20Because%20each%20row%20of%20%24A%24%20forms%20a%20valid%20probability%20distribution%20(%24%5Csum_%7Bj%3D1%7D%5EM%20A_%7Bij%7D%20%3D%201%24%20with%20%24A_%7Bij%7D%20%5Cge%200%24)%2C%20each%20output%20vector%20%24y_i%20%5Cin%20%5Cmathbb%7BR%7D%5E%7Bd_v%7D%24%20is%20a%20strictly%20**convex%20combination**%20of%20all%20value%20vectors%20%24v_1%2C%20%5Cdots%2C%20v_M%24.%0A%0A%20%20%20%20%23%23%23%202.%20The%20Variance%20Dilemma%3A%20Why%20Scale%20by%20%241%2F%5Csqrt%7Bd_k%7D%24%3F%0A%0A%20%20%20%20The%20scaling%20factor%20%24%5Cfrac%7B1%7D%7B%5Csqrt%7Bd_k%7D%7D%24%20is%20not%20an%20empirical%20heuristic%3B%20it%20is%20a%20mathematically%20required%20normalization%20to%20prevent%20softmax%20saturation.%0A%0A%20%20%20%20Consider%20a%20single%20query%20vector%20%24q%20%5Cin%20%5Cmathbb%7BR%7D%5E%7Bd_k%7D%24%20and%20key%20vector%20%24k%20%5Cin%20%5Cmathbb%7BR%7D%5E%7Bd_k%7D%24.%20Assume%20their%20components%20%24q_i%24%20and%20%24k_i%24%20are%20independent%2C%20identically%20distributed%20random%20variables%20with%20zero%20mean%20and%20unit%20variance%3A%0A%0A%20%20%20%20%24%24%5Cmathbb%7BE%7D%5Bq_i%5D%20%3D%20%5Cmathbb%7BE%7D%5Bk_i%5D%20%3D%200%2C%20%5Cqquad%20%5Coperatorname%7BVar%7D(q_i)%20%3D%20%5Coperatorname%7BVar%7D(k_i)%20%3D%201%24%24%0A%0A%20%20%20%20The%20unscaled%20dot%20product%20is%20the%20sum%20of%20%24d_k%24%20independent%20random%20products%3A%0A%0A%20%20%20%20%24%24z%20%3D%20q%5E%5Ctop%20k%20%3D%20%5Csum_%7Bi%3D1%7D%5E%7Bd_k%7D%20q_i%20k_i%24%24%0A%0A%20%20%20%20Evaluating%20the%20expectation%20and%20variance%20of%20%24z%24%3A%0A%0A%20%20%20%20%24%24%5Cmathbb%7BE%7D%5Bz%5D%20%3D%20%5Csum_%7Bi%3D1%7D%5E%7Bd_k%7D%20%5Cmathbb%7BE%7D%5Bq_i%5D%20%5Cmathbb%7BE%7D%5Bk_i%5D%20%3D%200%24%24%0A%0A%20%20%20%20By%20the%20product%20rule%20of%20variances%20for%20independent%20zero-mean%20variables%3A%0A%0A%20%20%20%20%24%24%5Coperatorname%7BVar%7D(q_i%20k_i)%20%3D%20%5Coperatorname%7BVar%7D(q_i)%5Coperatorname%7BVar%7D(k_i)%20%2B%20%5Coperatorname%7BVar%7D(q_i)%5Cmathbb%7BE%7D%5Bk_i%5D%5E2%20%2B%20%5Coperatorname%7BVar%7D(k_i)%5Cmathbb%7BE%7D%5Bq_i%5D%5E2%20%3D%201%20%5Ccdot%201%20%2B%200%20%2B%200%20%3D%201%24%24%0A%0A%20%20%20%20Summing%20across%20all%20%24d_k%24%20dimensions%3A%0A%0A%20%20%20%20%24%24%5Coperatorname%7BVar%7D(z)%20%3D%20%5Csum_%7Bi%3D1%7D%5E%7Bd_k%7D%20%5Coperatorname%7BVar%7D(q_i%20k_i)%20%3D%20d_k%2C%20%5Cqquad%20%5Csigma(z)%20%3D%20%5Csqrt%7Bd_k%7D%24%24%0A%0A%20%20%20%20%23%23%23%23%20Softmax%20Saturation%20and%20Vanishing%20Gradients%3A%0A%20%20%20%20As%20projection%20dimensionality%20%24d_k%24%20increases%20(e.g.%2C%20%24d_k%20%3D%2064%24%20in%20Base%20Transformers%2C%20%24d_k%20%3D%20128%24%20in%20LLaMA-3)%2C%20the%20standard%20deviation%20of%20raw%20dot%20products%20grows%20to%20%24%5Csqrt%7B64%7D%20%3D%208%24%20or%20%24%5Csqrt%7B128%7D%20%5Capprox%2011.3%24.%20The%20dot%20product%20values%20%24z%24%20routinely%20reach%20magnitudes%20in%20excess%20of%20%24%5Cpm%2025%24.%0A%0A%20%20%20%20Recall%20the%20Jacobian%20derivative%20of%20the%20softmax%20function%20with%20respect%20to%20its%20inputs%3A%0A%0A%20%20%20%20%24%24%5Cfrac%7B%5Cpartial%20A_i%7D%7B%5Cpartial%20S_j%7D%20%3D%20A_i%20(%5Cdelta_%7Bij%7D%20-%20A_j)%24%24%0A%0A%20%20%20%20When%20input%20magnitudes%20are%20large%2C%20the%20softmax%20output%20rapidly%20polarizes%20into%20a%20near-one-hot%20distribution%20(%24A_%7B%5Cmax%7D%20%5Capprox%201%24%2C%20all%20other%20%24A_j%20%5Capprox%200%24).%20In%20this%20regime%3A%0A%20%20%20%20-%20For%20the%20winning%20index%3A%20%24%5Cfrac%7B%5Cpartial%20A_i%7D%7B%5Cpartial%20S_i%7D%20%3D%201%20%5Ccdot%20(1%20-%201)%20%3D%200%24%0A%20%20%20%20-%20For%20all%20losing%20indices%3A%20%24%5Cfrac%7B%5Cpartial%20A_j%7D%7B%5Cpartial%20S_j%7D%20%3D%200%20%5Ccdot%20(1%20-%200)%20%3D%200%24%0A%0A%20%20%20%20The%20Jacobian%20vanishes%20completely%2C%20causing%20gradients%20to%20freeze%20and%20halting%20optimization.%20Dividing%20by%20%24%5Csqrt%7Bd_k%7D%24%20normalizes%20the%20variance%3A%0A%0A%20%20%20%20%24%24%5Coperatorname%7BVar%7D%5Cleft(%5Cfrac%7Bq%5E%5Ctop%20k%7D%7B%5Csqrt%7Bd_k%7D%7D%5Cright)%20%3D%20%5Cfrac%7B1%7D%7Bd_k%7D%20%5Coperatorname%7BVar%7D(q%5E%5Ctop%20k)%20%3D%20%5Cfrac%7Bd_k%7D%7Bd_k%7D%20%3D%201%24%24%0A%0A%20%20%20%20This%20maintains%20unit%20variance%20across%20all%20hidden%20dimensions%2C%20preserving%20active%20gradient%20propagation.%0A%0A%20%20%20%20%23%23%23%203.%20Backpropagation%20and%20Analytic%20Gradients%0A%0A%20%20%20%20During%20backpropagation%2C%20the%20loss%20gradient%20%24%5Cfrac%7B%5Cpartial%20%5Cmathcal%7BL%7D%7D%7B%5Cpartial%20Y%7D%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BN%20%5Ctimes%20d_v%7D%24%20flows%20backward%20through%20the%20attention%20layer%3A%0A%0A%20%20%20%20%24%24%5Cfrac%7B%5Cpartial%20%5Cmathcal%7BL%7D%7D%7B%5Cpartial%20V%7D%20%3D%20A%5E%5Ctop%20%5Cfrac%7B%5Cpartial%20%5Cmathcal%7BL%7D%7D%7B%5Cpartial%20Y%7D%24%24%0A%0A%20%20%20%20Let%20%24%5Cbar%7BA%7D%20%3D%20%5Cfrac%7B%5Cpartial%20%5Cmathcal%7BL%7D%7D%7B%5Cpartial%20Y%7D%20V%5E%5Ctop%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BN%20%5Ctimes%20M%7D%24.%20Using%20the%20vector-Jacobian%20product%20for%20softmax%3A%0A%0A%20%20%20%20%24%24%5Cfrac%7B%5Cpartial%20%5Cmathcal%7BL%7D%7D%7B%5Cpartial%20S_%7Bij%7D%7D%20%3D%20A_%7Bij%7D%20%5Cleft(%20%5Cbar%7BA%7D_%7Bij%7D%20-%20%5Csum_%7Bk%3D1%7D%5EM%20%5Cbar%7BA%7D_%7Bik%7D%20A_%7Bik%7D%20%5Cright)%24%24%0A%0A%20%20%20%20In%20matrix%20notation%2C%20where%20%24%5Cmathbf%7B1%7D%24%20is%20an%20all-ones%20column%20vector%3A%0A%0A%20%20%20%20%24%24%5Cfrac%7B%5Cpartial%20%5Cmathcal%7BL%7D%7D%7B%5Cpartial%20S%7D%20%3D%20A%20%5Codot%20%5Cleft(%20%5Cbar%7BA%7D%20-%20((%5Cbar%7BA%7D%20%5Codot%20A)%5Cmathbf%7B1%7D)%5Cmathbf%7B1%7D%5E%5Ctop%20%5Cright)%24%24%0A%0A%20%20%20%20Finally%2C%20applying%20the%20chain%20rule%20to%20the%20scaled%20projections%3A%0A%0A%20%20%20%20%24%24%5Cfrac%7B%5Cpartial%20%5Cmathcal%7BL%7D%7D%7B%5Cpartial%20Q%7D%20%3D%20%5Cfrac%7B1%7D%7B%5Csqrt%7Bd_k%7D%7D%20%5Cfrac%7B%5Cpartial%20%5Cmathcal%7BL%7D%7D%7B%5Cpartial%20S%7D%20K%2C%20%5Cqquad%20%5Cfrac%7B%5Cpartial%20%5Cmathcal%7BL%7D%7D%7B%5Cpartial%20K%7D%20%3D%20%5Cfrac%7B1%7D%7B%5Csqrt%7Bd_k%7D%7D%20%5Cleft(%5Cfrac%7B%5Cpartial%20%5Cmathcal%7BL%7D%7D%7B%5Cpartial%20S%7D%5Cright)%5E%5Ctop%20Q%24%24%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(go%2C%20make_subplots%2C%20mo%2C%20np)%3A%0A%20%20%20%20%23%20Demonstrate%20linguistic%20self-attention%20on%20an%20illustrative%20resolved%20pronoun%20sentence%0A%20%20%20%20sentence_tokens%20%3D%20%5B%22The%22%2C%20%22animal%22%2C%20%22didn't%22%2C%20%22cross%22%2C%20%22the%22%2C%20%22street%22%2C%20%22because%22%2C%20%22it%22%2C%20%22was%22%2C%20%22too%22%2C%20%22tired%22%5D%0A%20%20%20%20seq_len%20%3D%20len(sentence_tokens)%0A%0A%20%20%20%20%23%20Synthetic%20semantic%20embedding%20projections%20designed%20to%20simulate%20realistic%20coreference%20resolution%0A%20%20%20%20np.random.seed(1337)%0A%20%20%20%20d_k_demo%20%3D%2032%0A%0A%20%20%20%20%23%20Base%20embeddings%0A%20%20%20%20emb_matrix%20%3D%20np.random.normal(0%2C%200.5%2C%20(seq_len%2C%20d_k_demo))%0A%20%20%20%20%23%20Make%20%22animal%22%20and%20%22it%22%20share%20strong%20semantic%20query-key%20alignment%0A%20%20%20%20idx_animal%20%3D%20sentence_tokens.index(%22animal%22)%0A%20%20%20%20idx_it%20%3D%20sentence_tokens.index(%22it%22)%0A%20%20%20%20idx_street%20%3D%20sentence_tokens.index(%22street%22)%0A%20%20%20%20idx_tired%20%3D%20sentence_tokens.index(%22tired%22)%0A%0A%20%20%20%20%23%20Inject%20semantic%20correlation%0A%20%20%20%20emb_matrix%5Bidx_it%5D%20%3D%200.7%20*%20emb_matrix%5Bidx_animal%5D%20%2B%200.3%20*%20emb_matrix%5Bidx_tired%5D%20%2B%200.1%20*%20np.random.randn(d_k_demo)%0A%20%20%20%20emb_matrix%5Bidx_tired%5D%20%2B%3D%200.4%20*%20emb_matrix%5Bidx_animal%5D%0A%0A%20%20%20%20%23%20Attention%20scores%0A%20%20%20%20raw_dot_products%20%3D%20np.dot(emb_matrix%2C%20emb_matrix.T)%0A%20%20%20%20scaled_scores%20%3D%20raw_dot_products%20%2F%20np.sqrt(d_k_demo)%0A%0A%20%20%20%20%23%20Numerically%20stable%20softmax%0A%20%20%20%20exp_scaled%20%3D%20np.exp(scaled_scores%20-%20np.max(scaled_scores%2C%20axis%3D-1%2C%20keepdims%3DTrue))%0A%20%20%20%20attention_matrix%20%3D%20exp_scaled%20%2F%20np.sum(exp_scaled%2C%20axis%3D-1%2C%20keepdims%3DTrue)%0A%0A%20%20%20%20%23%20Softmax%20gradient%20simulation%3A%20Compare%20gradient%20magnitude%20as%20d_k%20scales%20from%204%20to%20512%0A%20%20%20%20dim_grid%20%3D%20np.array(%5B4%2C%208%2C%2016%2C%2032%2C%2064%2C%20128%2C%20256%2C%20512%5D)%0A%20%20%20%20n_trials%20%3D%20200%0A%0A%20%20%20%20grad_norm_unscaled%20%3D%20%5B%5D%0A%20%20%20%20grad_norm_scaled%20%3D%20%5B%5D%0A%0A%20%20%20%20for%20d%20in%20dim_grid%3A%0A%20%20%20%20%20%20%20%20%23%20Generate%20independent%20Gaussian%20queries%20and%20keys%0A%20%20%20%20%20%20%20%20q_samples%20%3D%20np.random.normal(0%2C%201%2C%20(n_trials%2C%20d))%0A%20%20%20%20%20%20%20%20k_samples%20%3D%20np.random.normal(0%2C%201%2C%20(n_trials%2C%2010%2C%20d))%20%20%23%2010%20keys%0A%0A%20%20%20%20%20%20%20%20%23%20Unscaled%20dot%20products%0A%20%20%20%20%20%20%20%20unscaled_z%20%3D%20np.einsum(%22td%2Ctkd-%3Etk%22%2C%20q_samples%2C%20k_samples)%0A%20%20%20%20%20%20%20%20%23%20Scaled%20dot%20products%0A%20%20%20%20%20%20%20%20scaled_z%20%3D%20unscaled_z%20%2F%20np.sqrt(d)%0A%0A%20%20%20%20%20%20%20%20%23%20Softmax%20computation%0A%20%20%20%20%20%20%20%20p_unscaled%20%3D%20np.exp(unscaled_z%20-%20np.max(unscaled_z%2C%20axis%3D1%2C%20keepdims%3DTrue))%0A%20%20%20%20%20%20%20%20p_unscaled%20%2F%3D%20np.sum(p_unscaled%2C%20axis%3D1%2C%20keepdims%3DTrue)%0A%0A%20%20%20%20%20%20%20%20p_scaled%20%3D%20np.exp(scaled_z%20-%20np.max(scaled_z%2C%20axis%3D1%2C%20keepdims%3DTrue))%0A%20%20%20%20%20%20%20%20p_scaled%20%2F%3D%20np.sum(p_scaled%2C%20axis%3D1%2C%20keepdims%3DTrue)%0A%0A%20%20%20%20%20%20%20%20%23%20Softmax%20diagonal%20Jacobian%20norm%3A%20mean%20of%20p_i%20*%20(1%20-%20p_i)%0A%20%20%20%20%20%20%20%20grad_unscaled_mean%20%3D%20np.mean(p_unscaled%20*%20(1.0%20-%20p_unscaled))%0A%20%20%20%20%20%20%20%20grad_scaled_mean%20%3D%20np.mean(p_scaled%20*%20(1.0%20-%20p_scaled))%0A%0A%20%20%20%20%20%20%20%20grad_norm_unscaled.append(grad_unscaled_mean)%0A%20%20%20%20%20%20%20%20grad_norm_scaled.append(grad_scaled_mean)%0A%0A%20%20%20%20fig%20%3D%20make_subplots(%0A%20%20%20%20%20%20%20%20rows%3D1%2C%0A%20%20%20%20%20%20%20%20cols%3D2%2C%0A%20%20%20%20%20%20%20%20subplot_titles%3D%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%3Cb%3ESelf-Attention%20Map%20A%20%3D%20Softmax(QK%5ET%20%2F%20sqrt(dk))%3C%2Fb%3E%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%3Cb%3ESoftmax%20Gradient%20Magnitude%20vs%20Key%20Dimension%20dk%3C%2Fb%3E%22%2C%0A%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%20%20%20%20horizontal_spacing%3D0.14%2C%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Panel%201%3A%20Attention%20Matrix%20Heatmap%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Heatmap(%0A%20%20%20%20%20%20%20%20%20%20%20%20z%3Dattention_matrix%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dsentence_tokens%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dsentence_tokens%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20colorscale%3D%22Blues%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20colorbar%3Ddict(title%3D%22Attention%20Weight%22%2C%20x%3D0.42)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20hoverongaps%3DFalse%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D1%2C%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Panel%202%3A%20Vanishing%20gradient%20curves%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Ddim_grid%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dgrad_norm_scaled%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%2Bmarkers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%231D4ED8%22%2C%20width%3D2.5)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20marker%3Ddict(size%3D8)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Scaled%20Attention%20(1%2Fsqrt(dk))%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Ddim_grid%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dgrad_norm_unscaled%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%2Bmarkers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%23DC2626%22%2C%20width%3D2.5%2C%20dash%3D%22dash%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20marker%3Ddict(size%3D8)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Unscaled%20Attention%20(Vanishing%20Gradient)%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%0A%20%20%20%20fig.update_xaxes(title_text%3D%22Key%20Token%22%2C%20row%3D1%2C%20col%3D1)%0A%20%20%20%20fig.update_yaxes(title_text%3D%22Query%20Token%22%2C%20autorange%3D%22reversed%22%2C%20row%3D1%2C%20col%3D1)%0A%20%20%20%20fig.update_xaxes(title_text%3D%22Key%20Projection%20Dimension%20dk%22%2C%20type%3D%22log%22%2C%20row%3D1%2C%20col%3D2)%0A%20%20%20%20fig.update_yaxes(title_text%3D%22Mean%20Softmax%20Gradient%20E%5Bp(1-p)%5D%22%2C%20range%3D%5B0%2C%200.12%5D%2C%20row%3D1%2C%20col%3D2)%0A%0A%20%20%20%20fig.update_layout(%0A%20%20%20%20%20%20%20%20template%3D%22plotly_white%22%2C%0A%20%20%20%20%20%20%20%20height%3D520%2C%0A%20%20%20%20%20%20%20%20margin%3Ddict(l%3D40%2C%20r%3D40%2C%20t%3D70%2C%20b%3D50)%2C%0A%20%20%20%20%20%20%20%20legend%3Ddict(orientation%3D%22h%22%2C%20yanchor%3D%22bottom%22%2C%20y%3D-0.25%2C%20xanchor%3D%22center%22%2C%20x%3D0.72)%2C%0A%20%20%20%20)%0A%0A%20%20%20%20viz%20%3D%20mo.ui.plotly(fig)%0A%20%20%20%20return%20dim_grid%2C%20grad_norm_scaled%2C%20grad_norm_unscaled%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(dim_grid%2C%20grad_norm_scaled%2C%20grad_norm_unscaled%2C%20mo%2C%20np%2C%20pd)%3A%0A%20%20%20%20%23%20Vectorized%20NumPy%20implementation%20of%20Scaled%20Dot-Product%20Attention%20with%20Analytic%20Backpropagation%0A%20%20%20%20def%20scaled_dot_product_attention_np(Q%2C%20K%2C%20V%2C%20mask%3DNone)%3A%0A%20%20%20%20%20%20%20%20%22%22%22Pure%20NumPy%20implementation%20of%20Scaled%20Dot-Product%20Attention.%0A%0A%20%20%20%20%20%20%20%20Args%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20Q%3A%20Query%20tensor%20of%20shape%20(...%2C%20N%2C%20d_k)%0A%20%20%20%20%20%20%20%20%20%20%20%20K%3A%20Key%20tensor%20of%20shape%20(...%2C%20M%2C%20d_k)%0A%20%20%20%20%20%20%20%20%20%20%20%20V%3A%20Value%20tensor%20of%20shape%20(...%2C%20M%2C%20d_v)%0A%20%20%20%20%20%20%20%20%20%20%20%20mask%3A%20Optional%20boolean%20mask%20of%20shape%20(...%2C%20N%2C%20M)%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20d_k%20%3D%20Q.shape%5B-1%5D%0A%20%20%20%20%20%20%20%20scores%20%3D%20np.matmul(Q%2C%20np.swapaxes(K%2C%20-1%2C%20-2))%20%2F%20np.sqrt(d_k)%0A%0A%20%20%20%20%20%20%20%20if%20mask%20is%20not%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20scores%20%3D%20np.where(mask%2C%20scores%2C%20-1e9)%0A%0A%20%20%20%20%20%20%20%20%23%20Numerically%20stable%20softmax%0A%20%20%20%20%20%20%20%20exp_s%20%3D%20np.exp(scores%20-%20np.max(scores%2C%20axis%3D-1%2C%20keepdims%3DTrue))%0A%20%20%20%20%20%20%20%20A%20%3D%20exp_s%20%2F%20np.sum(exp_s%2C%20axis%3D-1%2C%20keepdims%3DTrue)%0A%20%20%20%20%20%20%20%20Y%20%3D%20np.matmul(A%2C%20V)%0A%20%20%20%20%20%20%20%20return%20Y%2C%20A%0A%0A%20%20%20%20%23%20Numerical%20verification%20of%20exact%20convex%20combination%20and%20row-sum%20properties%0A%20%20%20%20np.random.seed(42)%0A%20%20%20%20_N%2C%20_M%2C%20_dk%2C%20_dv%20%3D%205%2C%206%2C%2016%2C%208%0A%20%20%20%20_Q%20%3D%20np.random.randn(_N%2C%20_dk)%0A%20%20%20%20_K%20%3D%20np.random.randn(_M%2C%20_dk)%0A%20%20%20%20_V%20%3D%20np.random.randn(_M%2C%20_dv)%0A%0A%20%20%20%20_Y%2C%20_A%20%3D%20scaled_dot_product_attention_np(_Q%2C%20_K%2C%20_V)%0A%0A%20%20%20%20%23%20Check%20row%20sums%20equal%201.0%0A%20%20%20%20_row_sums%20%3D%20np.sum(_A%2C%20axis%3D-1)%0A%20%20%20%20%23%20Check%20output%20range%20bounded%20by%20convex%20hull%20of%20V%0A%20%20%20%20_v_min_norm%20%3D%20np.min(np.linalg.norm(_V%2C%20axis%3D-1))%0A%20%20%20%20_v_max_norm%20%3D%20np.max(np.linalg.norm(_V%2C%20axis%3D-1))%0A%20%20%20%20_y_norms%20%3D%20np.linalg.norm(_Y%2C%20axis%3D-1)%0A%0A%20%20%20%20df_properties%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Mathematical_Property%22%3A%20%22Row%20Sum%20Normalization%20(sum_j%20A_ij%20%3D%201)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Theoretical_Value%22%3A%20%221.000000%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Empirical_Min%22%3A%20f%22%7Bnp.min(_row_sums)%3A.6f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Empirical_Max%22%3A%20f%22%7Bnp.max(_row_sums)%3A.6f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Verification_Status%22%3A%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Exact%20Match%22%20if%20np.allclose(_row_sums%2C%201.0%2C%20atol%3D1e-6)%20else%20%22Discrepancy%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Mathematical_Property%22%3A%20%22Non-Negativity%20(A_ij%20%3E%3D%200)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Theoretical_Value%22%3A%20%22%3E%3D%200.0%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Empirical_Min%22%3A%20f%22%7Bnp.min(_A)%3A.6f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Empirical_Max%22%3A%20f%22%7Bnp.max(_A)%3A.6f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Verification_Status%22%3A%20%22Strictly%20Non-Negative%22%20if%20np.all(_A%20%3E%3D%200.0)%20else%20%22Negative%20Entry%20Found%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Mathematical_Property%22%3A%20%22Convex%20Hull%20Norm%20Boundedness%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Theoretical_Value%22%3A%20f%22%5B%7B_v_min_norm%3A.2f%7D%2C%20%7B_v_max_norm%3A.2f%7D%5D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Empirical_Min%22%3A%20f%22%7Bnp.min(_y_norms)%3A.2f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Empirical_Max%22%3A%20f%22%7Bnp.max(_y_norms)%3A.2f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Verification_Status%22%3A%20%22Within%20Convex%20Hull%22%20if%20np.max(_y_norms)%20%3C%3D%20_v_max_norm%20*%201.01%20else%20%22Exceeded%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Example%202%3A%20Softmax%20Vanishing%20Gradient%20Audit%20Table%0A%20%20%20%20gradient_records%20%3D%20%5B%5D%0A%20%20%20%20for%20_idx%2C%20_d%20in%20enumerate(dim_grid)%3A%0A%20%20%20%20%20%20%20%20_ratio%20%3D%20grad_norm_scaled%5B_idx%5D%20%2F%20max(grad_norm_unscaled%5B_idx%5D%2C%201e-12)%0A%20%20%20%20%20%20%20%20gradient_records.append(%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Projection_Dimension_dk%22%3A%20int(_d)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Scaled_Grad_Norm%22%3A%20f%22%7Bgrad_norm_scaled%5B_idx%5D%3A.5f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Unscaled_Grad_Norm%22%3A%20f%22%7Bgrad_norm_unscaled%5B_idx%5D%3A.5f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Gradient_Ratio%20(Scaled%20%2F%20Unscaled)%22%3A%20f%22%7B_ratio%3A.1f%7Dx%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Regime%22%3A%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Stable%20Gradient%20Flow%22%20if%20grad_norm_unscaled%5B_idx%5D%20%3E%200.03%20else%20%22Severe%20Softmax%20Saturation%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20df_gradient_audit%20%3D%20pd.DataFrame(gradient_records)%0A%0A%20%20%20%20%23%20Example%203%3A%20Contextual%20Representation%20Shift%20(Polysemous%20Word%20Disambiguation)%0A%20%20%20%20%23%20Target%20word%3A%20%22bank%22%20in%20two%20distinct%20sentences%0A%20%20%20%20%23%20Sentence%201%3A%20%22deposit%20cash%20at%20river%20bank%22%20vs%20Sentence%202%3A%20%22deposit%20cash%20at%20investment%20bank%22%0A%20%20%20%20vocab%20%3D%20%5B%22deposit%22%2C%20%22cash%22%2C%20%22at%22%2C%20%22river%22%2C%20%22investment%22%2C%20%22bank%22%5D%0A%20%20%20%20v_dim%20%3D%208%0A%0A%20%20%20%20%23%20Semantic%20prototypes%0A%20%20%20%20np.random.seed(42)%0A%20%20%20%20semantic_bases%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22deposit%22%3A%20np.array(%5B2.0%2C%200.1%2C%200.0%2C%200.0%2C%201.5%2C%200.0%2C%200.0%2C%200.5%5D)%2C%0A%20%20%20%20%20%20%20%20%22cash%22%3A%20np.array(%5B2.5%2C%200.0%2C%200.0%2C%200.0%2C%202.0%2C%200.0%2C%200.0%2C%200.2%5D)%2C%0A%20%20%20%20%20%20%20%20%22at%22%3A%20np.array(%5B0.0%2C%200.0%2C%200.1%2C%200.0%2C%200.0%2C%200.0%2C%200.1%2C%200.0%5D)%2C%0A%20%20%20%20%20%20%20%20%22river%22%3A%20np.array(%5B0.0%2C%203.0%2C%202.5%2C%202.0%2C%200.0%2C%200.0%2C%200.0%2C%200.0%5D)%2C%0A%20%20%20%20%20%20%20%20%22investment%22%3A%20np.array(%5B3.0%2C%200.0%2C%200.0%2C%200.0%2C%203.0%2C%201.5%2C%200.0%2C%201.0%5D)%2C%0A%20%20%20%20%20%20%20%20%23%20Polysemous%20uncontextualized%20static%20embedding%20(mixture%20of%20finance%20and%20geography)%0A%20%20%20%20%20%20%20%20%22bank%22%3A%20np.array(%5B1.5%2C%201.5%2C%201.0%2C%201.0%2C%201.5%2C%200.5%2C%200.0%2C%200.5%5D)%2C%0A%20%20%20%20%7D%0A%0A%20%20%20%20%23%20Sentence%20A%3A%20river%20bank%0A%20%20%20%20sent_A%20%3D%20%5B%22deposit%22%2C%20%22cash%22%2C%20%22at%22%2C%20%22river%22%2C%20%22bank%22%5D%0A%20%20%20%20X_A%20%3D%20np.array(%5Bsemantic_bases%5Bw%5D%20for%20w%20in%20sent_A%5D)%0A%20%20%20%20Y_A%2C%20_%20%3D%20scaled_dot_product_attention_np(X_A%2C%20X_A%2C%20X_A)%0A%20%20%20%20bank_vec_A%20%3D%20Y_A%5Bsent_A.index(%22bank%22)%5D%0A%0A%20%20%20%20%23%20Sentence%20B%3A%20investment%20bank%0A%20%20%20%20sent_B%20%3D%20%5B%22deposit%22%2C%20%22cash%22%2C%20%22at%22%2C%20%22investment%22%2C%20%22bank%22%5D%0A%20%20%20%20X_B%20%3D%20np.array(%5Bsemantic_bases%5Bw%5D%20for%20w%20in%20sent_B%5D)%0A%20%20%20%20Y_B%2C%20_%20%3D%20scaled_dot_product_attention_np(X_B%2C%20X_B%2C%20X_B)%0A%20%20%20%20bank_vec_B%20%3D%20Y_B%5Bsent_B.index(%22bank%22)%5D%0A%0A%20%20%20%20def%20cosine_sim(u%2C%20v)%3A%0A%20%20%20%20%20%20%20%20return%20np.dot(u%2C%20v)%20%2F%20(np.linalg.norm(u)%20*%20np.linalg.norm(v))%0A%0A%20%20%20%20sim_static%20%3D%20cosine_sim(semantic_bases%5B%22bank%22%5D%2C%20semantic_bases%5B%22bank%22%5D)%20%20%23%201.0%0A%20%20%20%20sim_contextual%20%3D%20cosine_sim(bank_vec_A%2C%20bank_vec_B)%0A%20%20%20%20sim_bank_river%20%3D%20cosine_sim(bank_vec_A%2C%20semantic_bases%5B%22river%22%5D)%0A%20%20%20%20sim_bank_invest%20%3D%20cosine_sim(bank_vec_B%2C%20semantic_bases%5B%22investment%22%5D)%0A%0A%20%20%20%20df_disambiguation%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Comparison_Pair%22%3A%20%22Static%20bank%20vs%20Static%20bank%20(Pre-Attention)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Cosine_Similarity%22%3A%20f%22%7Bsim_static%3A.4f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Semantic_Interpretation%22%3A%20%22Single%20polysemous%20static%20vector%20cannot%20differentiate%20senses%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Comparison_Pair%22%3A%20%22Contextual%20bank%20(River)%20vs%20Contextual%20bank%20(Finance)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Cosine_Similarity%22%3A%20f%22%7Bsim_contextual%3A.4f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Semantic_Interpretation%22%3A%20%22Attention%20forces%20representations%20into%20distinct%20semantic%20regions%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Comparison_Pair%22%3A%20%22Contextual%20bank%20(River)%20vs%20Static%20'river'%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Cosine_Similarity%22%3A%20f%22%7Bsim_bank_river%3A.4f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Semantic_Interpretation%22%3A%20%22Substantial%20alignment%20with%20geographic%2Fhydrological%20context%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Comparison_Pair%22%3A%20%22Contextual%20bank%20(Finance)%20vs%20Static%20'investment'%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Cosine_Similarity%22%3A%20f%22%7Bsim_bank_invest%3A.4f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Semantic_Interpretation%22%3A%20%22Strong%20alignment%20with%20financial%20enterprise%20context%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%0A%20%20%20%20table_properties%20%3D%20mo.ui.table(df_properties)%0A%20%20%20%20table_gradient%20%3D%20mo.ui.table(df_gradient_audit)%0A%20%20%20%20table_disambig%20%3D%20mo.ui.table(df_disambiguation)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
819ee2648159d37263c389636778bf54