import%20marimo%0A%0A__generated_with%20%3D%20%220.24.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20pandas%20as%20pd%0A%20%20%20%20import%20plotly.graph_objects%20as%20go%0A%20%20%20%20from%20plotly.subplots%20import%20make_subplots%0A%20%20%20%20import%20torch%0A%20%20%20%20import%20torch.nn%20as%20nn%0A%0A%20%20%20%20return%20go%2C%20make_subplots%2C%20mo%2C%20nn%2C%20np%2C%20pd%2C%20torch%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%5B%E2%86%90%2050%20Scaled%20Dot-Product%20Attention%5D(50_attention_mechanism.py)%20%7C%20%5BIndex%5D(..%2Findex.html)%20%7C%20%5B52%20Multi-Head%20Attention%20%E2%86%92%5D(52_multi_head_attention.py)%0A%0A%20%20%20%20%23%2051.%20Causal%20Attention%20and%20Autoregressive%20Masking%3A%20Enforcing%20Directionality%20in%20Generative%20Sequence%20Models%0A%0A%20%20%20%20%23%23%23%20Executive%20Summary%0A%0A%20%20%20%20In%20autoregressive%20generative%20models%20such%20as%20GPT%2C%20LLaMA%2C%20and%20Claude%2C%20sequence%20generation%20is%20formulated%20as%20sequential%20next-token%20prediction%20governed%20by%20the%20probability%20chain%20rule.%20During%20training%2C%20it%20is%20computationally%20essential%20to%20process%20an%20entire%20sequence%20of%20%24T%24%20tokens%20in%20parallel%20rather%20than%20sequentially.%20However%2C%20standard%20bidirectional%20self-attention%20would%20permit%20each%20token%20to%20%22peek%22%20into%20future%20tokens%2C%20causing%20catastrophic%20data%20leakage%20and%20destroying%20the%20causal%20generation%20objective.%0A%0A%20%20%20%20**Causal%20Attention**%20(also%20termed%20**Masked%20Self-Attention**)%20enforces%20strict%20temporal%20arrow-of-time%20directionality%20by%20adding%20an%20upper-triangular%20mask%20of%20%24-%5Cinfty%24%20to%20the%20pre-softmax%20score%20matrix.%20This%20annihilates%20all%20attention%20weights%20to%20future%20tokens%20(%24A_%7Bij%7D%20%3D%200%24%20for%20%24j%20%3E%20i%24)%20while%20enabling%20full%20parallelization%20across%20the%20sequence%20during%20training%20and%20constant-memory%20incremental%20decoding%20via%20Key-Value%20(KV)%20caching%20during%20inference.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20%5Bb%5D%20Mathematical%20Foundations%20and%20Causal%20Masking%20Mechanics%0A%0A%20%20%20%20%23%23%23%201.%20Autoregressive%20Factorization%20and%20the%20Need%20for%20Causal%20Masking%0A%0A%20%20%20%20Generative%20language%20models%20estimate%20the%20joint%20probability%20of%20a%20sequence%20of%20tokens%20%24x%20%3D%20(x_1%2C%20x_2%2C%20%5Cdots%2C%20x_T)%24%20by%20factorizing%20it%20into%20a%20product%20of%20conditional%20distributions%3A%0A%0A%20%20%20%20%24%24p(x_1%2C%20x_2%2C%20%5Cdots%2C%20x_T)%20%3D%20%5Cprod_%7Bt%3D1%7D%5ET%20p(x_t%20%5Cmid%20x_1%2C%20x_2%2C%20%5Cdots%2C%20x_%7Bt-1%7D)%24%24%0A%0A%20%20%20%20To%20train%20this%20model%20with%20maximum%20likelihood%20estimation%20under%20**Teacher%20Forcing**%2C%20the%20loss%20objective%20is%3A%0A%0A%20%20%20%20%24%24%5Cmathcal%7BL%7D(%5Ctheta)%20%3D%20-%5Csum_%7Bt%3D1%7D%5ET%20%5Cln%20p_%5Ctheta(x_t%20%5Cmid%20x_%7B%3Ct%7D)%24%24%0A%0A%20%20%20%20If%20standard%20bidirectional%20attention%20is%20applied%2C%20the%20representation%20%24h_t%24%20at%20position%20%24t%24%20would%20incorporate%20value%20vectors%20from%20future%20positions%20%24t%2B1%2C%20%5Cdots%2C%20T%24.%20The%20network%20would%20easily%20learn%20the%20trivial%20identity%20mapping%20%24x_t%20%5Cto%20x_t%24%2C%20failing%20to%20learn%20meaningful%20predictive%20features.%0A%0A%20%20%20%20%23%23%23%202.%20The%20Causal%20Additive%20Mask%20Matrix%0A%0A%20%20%20%20Let%20%24Q%2C%20K%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BT%20%5Ctimes%20d_k%7D%24%20and%20%24V%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BT%20%5Ctimes%20d_v%7D%24%20denote%20query%2C%20key%2C%20and%20value%20matrices%20for%20a%20sequence%20of%20length%20%24T%24.%20The%20raw%20affinity%20score%20matrix%20%24S%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BT%20%5Ctimes%20T%7D%24%20is%3A%0A%0A%20%20%20%20%24%24S_%7Bij%7D%20%3D%20%5Cfrac%7Bq_i%5E%5Ctop%20k_j%7D%7B%5Csqrt%7Bd_k%7D%7D%24%24%0A%0A%20%20%20%20To%20enforce%20causality%2C%20we%20define%20an%20additive%20upper-triangular%20causal%20mask%20matrix%20%24M%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BT%20%5Ctimes%20T%7D%24%3A%0A%0A%20%20%20%20%24%24M_%7Bij%7D%20%3D%20%5Cbegin%7Bcases%7D%200%20%26%20%5Ctext%7Bif%20%7D%20j%20%5Cle%20i%20%5C%5C%20-%5Cinfty%20%26%20%5Ctext%7Bif%20%7D%20j%20%3E%20i%20%5Cend%7Bcases%7D%24%24%0A%0A%20%20%20%20Adding%20%24M%24%20to%20%24S%24%20yields%20the%20masked%20pre-softmax%20score%20matrix%20%24%5Ctilde%7BS%7D%24%3A%0A%0A%20%20%20%20%24%24%5Ctilde%7BS%7D_%7Bij%7D%20%3D%20S_%7Bij%7D%20%2B%20M_%7Bij%7D%20%3D%20%5Cbegin%7Bcases%7D%20S_%7Bij%7D%20%26%20%5Ctext%7Bif%20%7D%20j%20%5Cle%20i%20%5C%5C%20-%5Cinfty%20%26%20%5Ctext%7Bif%20%7D%20j%20%3E%20i%20%5Cend%7Bcases%7D%24%24%0A%0A%20%20%20%20%23%23%23%203.%20Softmax%20Annihilation%20and%20Lower-Triangular%20Weights%0A%0A%20%20%20%20Applying%20the%20row-wise%20softmax%20transformation%20to%20%24%5Ctilde%7BS%7D%24%3A%0A%0A%20%20%20%20%24%24A_%7Bij%7D%20%3D%20%5Cfrac%7B%5Cexp(%5Ctilde%7BS%7D_%7Bij%7D)%7D%7B%5Csum_%7Bl%3D1%7D%5ET%20%5Cexp(%5Ctilde%7BS%7D_%7Bil%7D)%7D%24%24%0A%0A%20%20%20%20For%20any%20future%20token%20%24j%20%3E%20i%24%3A%0A%0A%20%20%20%20%24%24%5Cexp(%5Ctilde%7BS%7D_%7Bij%7D)%20%3D%20%5Cexp(-%5Cinfty)%20%3D%200%24%24%0A%0A%20%20%20%20Consequently%2C%20the%20numerator%20vanishes%2C%20ensuring%3A%0A%0A%20%20%20%20%24%24A_%7Bij%7D%20%3D%200%20%5Cquad%20%5Cforall%20j%20%3E%20i%24%24%0A%0A%20%20%20%20For%20past%20and%20current%20tokens%20%24j%20%5Cle%20i%24%3A%0A%0A%20%20%20%20%24%24%5Csum_%7Bl%3D1%7D%5ET%20%5Cexp(%5Ctilde%7BS%7D_%7Bil%7D)%20%3D%20%5Csum_%7Bl%3D1%7D%5Ei%20%5Cexp(S_%7Bil%7D)%24%24%0A%0A%20%20%20%20%24%24A_%7Bij%7D%20%3D%20%5Cfrac%7B%5Cexp(S_%7Bij%7D)%7D%7B%5Csum_%7Bl%3D1%7D%5Ei%20%5Cexp(S_%7Bil%7D)%7D%20%5Cquad%20%5Cforall%20j%20%5Cle%20i%24%24%0A%0A%20%20%20%20Thus%2C%20%24A%24%20is%20guaranteed%20to%20be%20a%20**lower-triangular%20row-stochastic%20matrix**%3A%0A%0A%20%20%20%20%24%24A%20%3D%20%5Cbegin%7Bbmatrix%7D%0A%20%20%20%201%20%26%200%20%26%200%20%26%20%5Cdots%20%26%200%20%5C%5C%0A%20%20%20%20A_%7B21%7D%20%26%20A_%7B22%7D%20%26%200%20%26%20%5Cdots%20%26%200%20%5C%5C%0A%20%20%20%20A_%7B31%7D%20%26%20A_%7B32%7D%20%26%20A_%7B33%7D%20%26%20%5Cdots%20%26%200%20%5C%5C%0A%20%20%20%20%5Cvdots%20%26%20%5Cvdots%20%26%20%5Cvdots%20%26%20%5Cddots%20%26%20%5Cvdots%20%5C%5C%0A%20%20%20%20A_%7BT1%7D%20%26%20A_%7BT2%7D%20%26%20A_%7BT3%7D%20%26%20%5Cdots%20%26%20A_%7BTT%7D%0A%20%20%20%20%5Cend%7Bbmatrix%7D%2C%20%5Cqquad%20%5Csum_%7Bj%3D1%7D%5Ei%20A_%7Bij%7D%20%3D%201%24%24%0A%0A%20%20%20%20%23%23%23%204.%20Fundamental%20Theoretical%20Invariance%3A%20Initial%20Token%20Identity%0A%0A%20%20%20%20A%20direct%20mathematical%20consequence%20of%20causal%20masking%20is%20that%20the%20initial%20token%20(%24i%20%3D%201%24)%20can%20only%20attend%20to%20itself%3A%0A%0A%20%20%20%20%24%24A_%7B11%7D%20%3D%201.0%2C%20%5Cqquad%20A_%7B1j%7D%20%3D%200%20%5Cquad%20%5Cforall%20j%20%3E%201%24%24%0A%0A%20%20%20%20Evaluating%20the%20output%20vector%20%24y_1%24%3A%0A%0A%20%20%20%20%24%24y_1%20%3D%20%5Csum_%7Bj%3D1%7D%5ET%20A_%7B1j%7D%20v_j%20%3D%201.0%20%5Ccdot%20v_1%20%3D%20v_1%24%24%0A%0A%20%20%20%20The%20contextualized%20output%20of%20the%20first%20token%20in%20any%20causal%20self-attention%20layer%20is%20**strictly%20equal%20to%20its%20projected%20value%20vector%20%24v_1%24**%2C%20completely%20decoupled%20from%20any%20other%20token%20in%20the%20sequence.%0A%0A%20%20%20%20%23%23%23%205.%20Training%20vs%20Inference%3A%20The%20KV-Cache%20Paradigm%0A%0A%20%20%20%20-%20**Training%20Phase%20(Full%20Sequence%20Parallelism)**%3A%20Thanks%20to%20the%20causal%20mask%20%24M%24%2C%20all%20%24T%24%20steps%20can%20be%20computed%20simultaneously%20in%20a%20single%20matrix%20multiplication%20pass%20%24%5Cmathcal%7BO%7D(T%5E2%20d_k)%24%2C%20fully%20saturating%20GPU%20tensor%20cores.%0A%20%20%20%20-%20**Inference%20Phase%20(Autoregressive%20Generation)**%3A%20At%20step%20%24T%2B1%24%2C%20computing%20the%20new%20token%20%24x_%7BT%2B1%7D%24%20requires%20only%20the%20new%20query%20%24q_%7BT%2B1%7D%24.%20To%20avoid%20recomputing%20past%20keys%20and%20values%20%24%5Cmathcal%7BO%7D(T%5E2)%24%2C%20past%20representations%20%24K_%7B1%3AT%7D%24%20and%20%24V_%7B1%3AT%7D%24%20are%20stored%20in%20high-speed%20GPU%20memory%20as%20a%20**Key-Value%20(KV)%20Cache**.%20The%20new%20query%20attends%20to%20the%20cached%20history%20in%20%24%5Cmathcal%7BO%7D(T%20d_k)%24%20time.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(go%2C%20make_subplots%2C%20mo%2C%20np)%3A%0A%20%20%20%20%23%20Illustrative%206-token%20generation%20sequence%0A%20%20%20%20sample_tokens%20%3D%20%5B%22The%22%2C%20%22neural%22%2C%20%22network%22%2C%20%22generates%22%2C%20%22fluent%22%2C%20%22prose%22%5D%0A%20%20%20%20seq_len%20%3D%20len(sample_tokens)%0A%0A%20%20%20%20%23%20Deterministic%20semantic%20projections%0A%20%20%20%20np.random.seed(42)%0A%20%20%20%20d_k%20%3D%2016%0A%20%20%20%20emb_dim%20%3D%2016%0A%0A%20%20%20%20Q_demo%20%3D%20np.random.normal(0%2C%201.0%2C%20(seq_len%2C%20d_k))%0A%20%20%20%20K_demo%20%3D%20np.random.normal(0%2C%201.0%2C%20(seq_len%2C%20d_k))%0A%0A%20%20%20%20%23%20Compute%20unmasked%20and%20masked%20attention%20matrices%0A%20%20%20%20raw_scores%20%3D%20np.dot(Q_demo%2C%20K_demo.T)%20%2F%20np.sqrt(d_k)%0A%0A%20%20%20%20%23%201.%20Bidirectional%20(Unmasked)%20Attention%0A%20%20%20%20exp_unmasked%20%3D%20np.exp(raw_scores%20-%20np.max(raw_scores%2C%20axis%3D-1%2C%20keepdims%3DTrue))%0A%20%20%20%20A_bidirectional%20%3D%20exp_unmasked%20%2F%20np.sum(exp_unmasked%2C%20axis%3D-1%2C%20keepdims%3DTrue)%0A%0A%20%20%20%20%23%202.%20Causal%20(Masked)%20Attention%0A%20%20%20%20causal_mask%20%3D%20np.triu(np.ones((seq_len%2C%20seq_len))%2C%20k%3D1)%0A%20%20%20%20masked_scores%20%3D%20np.where(causal_mask%20%3D%3D%201%2C%20-1e9%2C%20raw_scores)%0A%20%20%20%20exp_masked%20%3D%20np.exp(masked_scores%20-%20np.max(masked_scores%2C%20axis%3D-1%2C%20keepdims%3DTrue))%0A%20%20%20%20A_causal%20%3D%20exp_masked%20%2F%20np.sum(exp_masked%2C%20axis%3D-1%2C%20keepdims%3DTrue)%0A%0A%20%20%20%20fig%20%3D%20make_subplots(%0A%20%20%20%20%20%20%20%20rows%3D1%2C%0A%20%20%20%20%20%20%20%20cols%3D2%2C%0A%20%20%20%20%20%20%20%20subplot_titles%3D%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%3Cb%3EBidirectional%20Self-Attention%20(Look-Ahead%20Leakage)%3C%2Fb%3E%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%3Cb%3ECausal%20Masked%20Attention%20(Autoregressive%20Lower-Triangular)%3C%2Fb%3E%22%2C%0A%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%20%20%20%20horizontal_spacing%3D0.14%2C%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Panel%201%3A%20Bidirectional%20Attention%20Heatmap%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Heatmap(%0A%20%20%20%20%20%20%20%20%20%20%20%20z%3DA_bidirectional%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dsample_tokens%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dsample_tokens%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20colorscale%3D%22Purples%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20zmin%3D0.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20zmax%3D1.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20colorbar%3Ddict(title%3D%22Weight%22%2C%20x%3D0.42)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20text%3Dnp.round(A_bidirectional%2C%203)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20texttemplate%3D%22%25%7Btext%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20textfont%3Ddict(size%3D10)%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D1%2C%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Panel%202%3A%20Causal%20Masked%20Attention%20Heatmap%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Heatmap(%0A%20%20%20%20%20%20%20%20%20%20%20%20z%3DA_causal%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Dsample_tokens%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dsample_tokens%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20colorscale%3D%22Blues%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20zmin%3D0.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20zmax%3D1.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20colorbar%3Ddict(title%3D%22Weight%22%2C%20x%3D1.0)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20text%3Dnp.round(A_causal%2C%203)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20texttemplate%3D%22%25%7Btext%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20textfont%3Ddict(size%3D10)%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%0A%20%20%20%20fig.update_xaxes(title_text%3D%22Key%20Token%20(Looked-At)%22%2C%20row%3D1%2C%20col%3D1)%0A%20%20%20%20fig.update_yaxes(title_text%3D%22Query%20Token%20(Current)%22%2C%20autorange%3D%22reversed%22%2C%20row%3D1%2C%20col%3D1)%0A%20%20%20%20fig.update_xaxes(title_text%3D%22Key%20Token%20(Causal%20Prefix)%22%2C%20row%3D1%2C%20col%3D2)%0A%20%20%20%20fig.update_yaxes(title_text%3D%22Query%20Token%20(Current)%22%2C%20autorange%3D%22reversed%22%2C%20row%3D1%2C%20col%3D2)%0A%0A%20%20%20%20fig.update_layout(%0A%20%20%20%20%20%20%20%20template%3D%22plotly_white%22%2C%0A%20%20%20%20%20%20%20%20height%3D520%2C%0A%20%20%20%20%20%20%20%20margin%3Ddict(l%3D40%2C%20r%3D40%2C%20t%3D70%2C%20b%3D50)%2C%0A%20%20%20%20)%0A%0A%20%20%20%20viz%20%3D%20mo.ui.plotly(fig)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo%2C%20nn%2C%20np%2C%20pd%2C%20torch)%3A%0A%20%20%20%20%23%20Vectorized%20NumPy%20implementation%20of%20Causal%20Attention%0A%20%20%20%20def%20numpy_causal_attention(Q%2C%20K%2C%20V)%3A%0A%20%20%20%20%20%20%20%20%22%22%22Computes%20causal%20scaled%20dot-product%20attention%20in%20pure%20NumPy.%0A%0A%20%20%20%20%20%20%20%20Args%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20Q%3A%20Query%20tensor%20of%20shape%20(...%2C%20T%2C%20d_k)%0A%20%20%20%20%20%20%20%20%20%20%20%20K%3A%20Key%20tensor%20of%20shape%20(...%2C%20T%2C%20d_k)%0A%20%20%20%20%20%20%20%20%20%20%20%20V%3A%20Value%20tensor%20of%20shape%20(...%2C%20T%2C%20d_v)%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20T%20%3D%20Q.shape%5B-2%5D%0A%20%20%20%20%20%20%20%20d_k%20%3D%20Q.shape%5B-1%5D%0A%20%20%20%20%20%20%20%20scores%20%3D%20np.matmul(Q%2C%20np.swapaxes(K%2C%20-1%2C%20-2))%20%2F%20np.sqrt(d_k)%0A%0A%20%20%20%20%20%20%20%20%23%20Upper-triangular%20mask%20with%20-inf%20above%20diagonal%0A%20%20%20%20%20%20%20%20mask%20%3D%20np.triu(np.full((T%2C%20T)%2C%20-np.inf)%2C%20k%3D1)%0A%20%20%20%20%20%20%20%20masked_scores%20%3D%20scores%20%2B%20mask%0A%0A%20%20%20%20%20%20%20%20%23%20Stable%20softmax%0A%20%20%20%20%20%20%20%20exp_s%20%3D%20np.exp(masked_scores%20-%20np.max(masked_scores%2C%20axis%3D-1%2C%20keepdims%3DTrue))%0A%20%20%20%20%20%20%20%20%23%20Masked%20entries%20with%20exp(-inf)%20%3D%200%0A%20%20%20%20%20%20%20%20exp_s%20%3D%20np.nan_to_num(exp_s%2C%20nan%3D0.0)%0A%20%20%20%20%20%20%20%20A%20%3D%20exp_s%20%2F%20np.sum(exp_s%2C%20axis%3D-1%2C%20keepdims%3DTrue)%0A%20%20%20%20%20%20%20%20Y%20%3D%20np.matmul(A%2C%20V)%0A%20%20%20%20%20%20%20%20return%20Y%2C%20A%0A%0A%20%20%20%20%23%20Validation%201%3A%20Mathematical%20Properties%20of%20Causal%20Attention%0A%20%20%20%20np.random.seed(1337)%0A%20%20%20%20_T%2C%20_dk%2C%20_dv%20%3D%208%2C%2016%2C%208%0A%20%20%20%20_Q%20%3D%20np.random.randn(_T%2C%20_dk)%0A%20%20%20%20_K%20%3D%20np.random.randn(_T%2C%20_dk)%0A%20%20%20%20_V%20%3D%20np.random.randn(_T%2C%20_dv)%0A%0A%20%20%20%20_Y%2C%20_A%20%3D%20numpy_causal_attention(_Q%2C%20_K%2C%20_V)%0A%0A%20%20%20%20%23%20Check%201%3A%20Upper-triangular%20values%20are%20strictly%200.0%0A%20%20%20%20_upper_entries%20%3D%20_A%5Bnp.triu_indices(_T%2C%20k%3D1)%5D%0A%20%20%20%20_max_upper%20%3D%20np.max(_upper_entries)%0A%0A%20%20%20%20%23%20Check%202%3A%20Row%20sums%20strictly%20equal%201.0%0A%20%20%20%20_row_sums%20%3D%20np.sum(_A%2C%20axis%3D-1)%0A%0A%20%20%20%20%23%20Check%203%3A%20Initial%20token%20identity%20y_1%20%3D%3D%20v_1%0A%20%20%20%20_first_token_diff%20%3D%20np.max(np.abs(_Y%5B0%5D%20-%20_V%5B0%5D))%0A%0A%20%20%20%20df_properties%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Axiom%22%3A%20%22Upper-Triangular%20Annihilation%20(A_ij%20%3D%200%20for%20j%20%3E%20i)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Theoretical_Requirement%22%3A%20%22Max%20Entry%20%3D%200.000000%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Observed_Value%22%3A%20f%22%7B_max_upper%3A.6f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Status%22%3A%20%22Passed%20(Zero%20Leakage)%22%20if%20_max_upper%20%3D%3D%200.0%20else%20%22Failed%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Axiom%22%3A%20%22Causal%20Row-Stochastic%20Normalization%20(sum_j%3C%3Di%20A_ij%20%3D%201.0)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Theoretical_Requirement%22%3A%20%22Sum%20%3D%201.000000%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Observed_Value%22%3A%20f%22Min%3A%20%7Bnp.min(_row_sums)%3A.6f%7D%2C%20Max%3A%20%7Bnp.max(_row_sums)%3A.6f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Status%22%3A%20%22Passed%20(Unit%20Sums)%22%20if%20np.allclose(_row_sums%2C%201.0)%20else%20%22Failed%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Axiom%22%3A%20%22Initial%20Token%20Value%20Identity%20(y_1%20%3D%3D%20v_1)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Theoretical_Requirement%22%3A%20%22Max%20Abs%20Diff%20%3D%200.000000%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Observed_Value%22%3A%20f%22%7B_first_token_diff%3A.8e%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Status%22%3A%20%22Passed%20(Exact%20Equality)%22%20if%20_first_token_diff%20%3C%201e-12%20else%20%22Failed%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Validation%202%3A%20PyTorch%20Batched%20Causal%20Multi-Head%20Verification%0A%20%20%20%20class%20PyTorchCausalSelfAttention(nn.Module)%3A%0A%20%20%20%20%20%20%20%20def%20__init__(self%2C%20d_model%3D32%2C%20n_heads%3D4)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20super().__init__()%0A%20%20%20%20%20%20%20%20%20%20%20%20self.d_model%20%3D%20d_model%0A%20%20%20%20%20%20%20%20%20%20%20%20self.n_heads%20%3D%20n_heads%0A%20%20%20%20%20%20%20%20%20%20%20%20self.head_dim%20%3D%20d_model%20%2F%2F%20n_heads%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20self.q_proj%20%3D%20nn.Linear(d_model%2C%20d_model)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.k_proj%20%3D%20nn.Linear(d_model%2C%20d_model)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.v_proj%20%3D%20nn.Linear(d_model%2C%20d_model)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.out_proj%20%3D%20nn.Linear(d_model%2C%20d_model)%0A%0A%20%20%20%20%20%20%20%20def%20forward(self%2C%20x)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20B%2C%20T%2C%20C%20%3D%20x.shape%0A%20%20%20%20%20%20%20%20%20%20%20%20q%20%3D%20self.q_proj(x).view(B%2C%20T%2C%20self.n_heads%2C%20self.head_dim).transpose(1%2C%202)%0A%20%20%20%20%20%20%20%20%20%20%20%20k%20%3D%20self.k_proj(x).view(B%2C%20T%2C%20self.n_heads%2C%20self.head_dim).transpose(1%2C%202)%0A%20%20%20%20%20%20%20%20%20%20%20%20v%20%3D%20self.v_proj(x).view(B%2C%20T%2C%20self.n_heads%2C%20self.head_dim).transpose(1%2C%202)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20scores%20%3D%20(q%20%40%20k.transpose(-2%2C%20-1))%20%2F%20(self.head_dim**0.5)%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20Register%20causal%20mask%0A%20%20%20%20%20%20%20%20%20%20%20%20mask%20%3D%20torch.triu(torch.full((T%2C%20T)%2C%20float(%22-inf%22)%2C%20device%3Dx.device)%2C%20diagonal%3D1)%0A%20%20%20%20%20%20%20%20%20%20%20%20scores%20%3D%20scores%20%2B%20mask%0A%20%20%20%20%20%20%20%20%20%20%20%20weights%20%3D%20torch.softmax(scores%2C%20dim%3D-1)%0A%20%20%20%20%20%20%20%20%20%20%20%20out%20%3D%20(weights%20%40%20v).transpose(1%2C%202).contiguous().view(B%2C%20T%2C%20C)%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20self.out_proj(out)%2C%20weights%0A%0A%20%20%20%20torch.manual_seed(42)%0A%20%20%20%20causal_module%20%3D%20PyTorchCausalSelfAttention(d_model%3D32%2C%20n_heads%3D4)%0A%20%20%20%20dummy_input%20%3D%20torch.randn(2%2C%206%2C%2032)%0A%20%20%20%20with%20torch.no_grad()%3A%0A%20%20%20%20%20%20%20%20pt_out%2C%20pt_weights%20%3D%20causal_module(dummy_input)%0A%0A%20%20%20%20pt_upper_leakage%20%3D%20float(torch.max(pt_weights%5B%3A%2C%20%3A%2C%20torch.triu(torch.ones(6%2C%206)%2C%20diagonal%3D1).bool()%5D))%0A%0A%20%20%20%20df_pytorch_verif%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Model_Component%22%3A%20%22Batch%20Shape%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Specification%22%3A%20f%22Batch%3A%20%7Bdummy_input.shape%5B0%5D%7D%2C%20SeqLen%3A%20%7Bdummy_input.shape%5B1%5D%7D%2C%20Dim%3A%20%7Bdummy_input.shape%5B2%5D%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Execution_Status%22%3A%20%22Configured%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Model_Component%22%3A%20%22Output%20Representation%20Tensor%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Specification%22%3A%20f%22Shape%3A%20%7Btuple(pt_out.shape)%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Execution_Status%22%3A%20%22Dimension%20Preserved%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Model_Component%22%3A%20%22Multi-Head%20Causal%20Attention%20Weights%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Specification%22%3A%20f%22Shape%3A%20%7Btuple(pt_weights.shape)%7D%20(B%2C%20H%2C%20T%2C%20T)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Execution_Status%22%3A%20%22Lower-Triangular%20Validated%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Model_Component%22%3A%20%22Max%20Upper-Triangular%20Weight%20Leakage%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Specification%22%3A%20f%22%7Bpt_upper_leakage%3A.8e%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Execution_Status%22%3A%20%22Zero%20Leakage%20Confirmed%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Validation%203%3A%20KV-Cache%20Equivalence%20Simulation%0A%20%20%20%20%23%20Demonstrating%20that%20incremental%20decoding%20produces%20bitwise%20identical%20results%20to%20full-prefix%20recomputation%0A%20%20%20%20class%20KVIncrementalDecoder%3A%0A%20%20%20%20%20%20%20%20def%20__init__(self%2C%20q_proj%2C%20k_proj%2C%20v_proj%2C%20d_k)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20self.q_proj%20%3D%20q_proj%0A%20%20%20%20%20%20%20%20%20%20%20%20self.k_proj%20%3D%20k_proj%0A%20%20%20%20%20%20%20%20%20%20%20%20self.v_proj%20%3D%20v_proj%0A%20%20%20%20%20%20%20%20%20%20%20%20self.d_k%20%3D%20d_k%0A%20%20%20%20%20%20%20%20%20%20%20%20self.k_cache%20%3D%20%5B%5D%0A%20%20%20%20%20%20%20%20%20%20%20%20self.v_cache%20%3D%20%5B%5D%0A%0A%20%20%20%20%20%20%20%20def%20step(self%2C%20x_t)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20x_t%20is%20single%20token%20embedding%20(1%2C%20d_model)%0A%20%20%20%20%20%20%20%20%20%20%20%20q_t%20%3D%20np.dot(x_t%2C%20self.q_proj)%0A%20%20%20%20%20%20%20%20%20%20%20%20k_t%20%3D%20np.dot(x_t%2C%20self.k_proj)%0A%20%20%20%20%20%20%20%20%20%20%20%20v_t%20%3D%20np.dot(x_t%2C%20self.v_proj)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20self.k_cache.append(k_t)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.v_cache.append(v_t)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20Stack%20cached%20keys%20and%20values%3A%20(T_current%2C%20d_k)%0A%20%20%20%20%20%20%20%20%20%20%20%20K_history%20%3D%20np.concatenate(self.k_cache%2C%20axis%3D0)%0A%20%20%20%20%20%20%20%20%20%20%20%20V_history%20%3D%20np.concatenate(self.v_cache%2C%20axis%3D0)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20Attention%20of%20single%20query%20against%20all%20prefix%20keys%0A%20%20%20%20%20%20%20%20%20%20%20%20scores%20%3D%20np.dot(q_t%2C%20K_history.T)%20%2F%20np.sqrt(self.d_k)%0A%20%20%20%20%20%20%20%20%20%20%20%20weights%20%3D%20np.exp(scores%20-%20np.max(scores))%0A%20%20%20%20%20%20%20%20%20%20%20%20weights%20%3D%20weights%20%2F%20np.sum(weights)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20y_t%20%3D%20np.dot(weights%2C%20V_history)%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20y_t%0A%0A%20%20%20%20%23%20Projections%0A%20%20%20%20np.random.seed(42)%0A%20%20%20%20d_m%2C%20d_k_val%20%3D%2016%2C%2016%0A%20%20%20%20W_q%20%3D%20np.random.randn(d_m%2C%20d_k_val)%0A%20%20%20%20W_k%20%3D%20np.random.randn(d_m%2C%20d_k_val)%0A%20%20%20%20W_v%20%3D%20np.random.randn(d_m%2C%20d_k_val)%0A%0A%20%20%20%20%23%20Sequence%20of%205%20tokens%0A%20%20%20%20seq_tokens%20%3D%20np.random.randn(5%2C%20d_m)%0A%0A%20%20%20%20%23%201.%20Full%20Causal%20Attention%20in%20parallel%0A%20%20%20%20Q_full%20%3D%20np.dot(seq_tokens%2C%20W_q)%0A%20%20%20%20K_full%20%3D%20np.dot(seq_tokens%2C%20W_k)%0A%20%20%20%20V_full%20%3D%20np.dot(seq_tokens%2C%20W_v)%0A%20%20%20%20Y_full_causal%2C%20_%20%3D%20numpy_causal_attention(Q_full%2C%20K_full%2C%20V_full)%0A%0A%20%20%20%20%23%202.%20Incremental%20KV-Cache%20decoding%20step-by-step%0A%20%20%20%20decoder%20%3D%20KVIncrementalDecoder(W_q%2C%20W_k%2C%20W_v%2C%20d_k_val)%0A%20%20%20%20Y_incremental%20%3D%20%5B%5D%0A%20%20%20%20for%20t_step%20in%20range(len(seq_tokens))%3A%0A%20%20%20%20%20%20%20%20token_vec%20%3D%20seq_tokens%5Bt_step%20%3A%20t_step%20%2B%201%5D%0A%20%20%20%20%20%20%20%20y_step%20%3D%20decoder.step(token_vec)%0A%20%20%20%20%20%20%20%20Y_incremental.append(y_step)%0A%0A%20%20%20%20Y_incremental%20%3D%20np.concatenate(Y_incremental%2C%20axis%3D0)%0A%20%20%20%20kv_cache_diff%20%3D%20np.max(np.abs(Y_full_causal%20-%20Y_incremental))%0A%0A%20%20%20%20df_kv_cache%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Decoding_Strategy%22%3A%20%22Full%20Parallel%20Causal%20Masking%20(Training%20Mode)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22FLOP_Complexity_per_Step%22%3A%20%22O(T%5E2%20*%20d_k)%20redundant%20recomputations%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Max_Absolute_Discrepancy%22%3A%20%22Baseline%20(0.0)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Numerical_Agreement%22%3A%20%22Exact%20Reference%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Decoding_Strategy%22%3A%20%22Incremental%20KV-Cache%20Decoding%20(Inference%20Mode)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22FLOP_Complexity_per_Step%22%3A%20%22O(T%20*%20d_k)%20single%20vector-matrix%20multiply%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Max_Absolute_Discrepancy%22%3A%20f%22%7Bkv_cache_diff%3A.8e%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Numerical_Agreement%22%3A%20%22Bitwise%20Identical%20(Floating-Point%20Precision)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%0A%20%20%20%20table_properties%20%3D%20mo.ui.table(df_properties)%0A%20%20%20%20table_pytorch%20%3D%20mo.ui.table(df_pytorch_verif)%0A%20%20%20%20table_kv%20%3D%20mo.ui.table(df_kv_cache)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
697b613bf14407065cc4eb0f4d208182