import%20marimo%0A%0A__generated_with%20%3D%20%220.24.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20pandas%20as%20pd%0A%20%20%20%20import%20plotly.graph_objects%20as%20go%0A%20%20%20%20from%20plotly.subplots%20import%20make_subplots%0A%20%20%20%20import%20torch%0A%20%20%20%20import%20torch.nn%20as%20nn%0A%0A%20%20%20%20return%20go%2C%20make_subplots%2C%20mo%2C%20nn%2C%20np%2C%20pd%2C%20torch%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%5B%E2%86%90%2051%20Causal%20Attention%5D(51_causal_attention.py)%20%7C%20%5BIndex%5D(..%2Findex.html)%20%7C%20%5B53%20LayerNorm%20vs%20RMSNorm%20%E2%86%92%5D(53_layernorm_vs_rmsnorm.py)%0A%0A%20%20%20%20%23%2052.%20Multi-Head%20Attention%3A%20Representation%20Subspaces%20and%20Parallel%20Projection%20Dynamics%0A%0A%20%20%20%20%23%23%23%20Executive%20Summary%0A%0A%20%20%20%20While%20single-head%20scaled%20dot-product%20attention%20constructs%20contextual%20embeddings%20via%20convex%20combinations%20of%20value%20vectors%2C%20it%20suffers%20from%20a%20fundamental%20mathematical%20bottleneck%3A%20all%20token%20interactions%20are%20compressed%20into%20a%20single%20probability%20distribution.%20If%20a%20token%20possesses%20simultaneous%20syntactic%2C%20semantic%2C%20and%20positional%20dependencies%20(e.g.%2C%20subject-verb%20agreement%2C%20coreference%20resolution%2C%20and%20adjacent%20bigram%20coupling)%2C%20a%20single%20attention%20head%20is%20forced%20to%20average%20across%20these%20conflicting%20signals%2C%20diluting%20its%20representational%20precision.%0A%0A%20%20%20%20**Multi-Head%20Attention%20(MHA)**%20(Vaswani%20et%20al.%2C%202017)%20resolves%20this%20constraint%20by%20linearly%20projecting%20Queries%2C%20Keys%2C%20and%20Values%20into%20%24h%24%20distinct%20lower-dimensional%20subspaces%20of%20dimension%20%24d_k%20%3D%20d_%7B%5Ctext%7Bmodel%7D%7D%20%2F%20h%24.%20Each%20head%20executes%20attention%20independently%20in%20parallel%2C%20enabling%20the%20network%20to%20jointly%20attend%20to%20information%20from%20disparate%20representation%20subspaces%20at%20disparate%20sequence%20positions.%20The%20individual%20head%20outputs%20are%20then%20concatenated%20and%20projected%20back%20into%20the%20model%20dimension%20via%20an%20output%20matrix%20%24W%5EO%24%2C%20perfectly%20preserving%20the%20overall%20computational%20FLOP%20budget.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20%5Bb%5D%20Mathematical%20Foundations%20and%20Subspace%20Projections%0A%0A%20%20%20%20%23%23%23%201.%20The%20Multi-Head%20Attention%20Equations%0A%0A%20%20%20%20Let%20%24X%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BN%20%5Ctimes%20d_%7B%5Ctext%7Bmodel%7D%7D%7D%24%20denote%20the%20input%20matrix%20of%20%24N%24%20token%20embeddings.%20For%20%24h%24%20attention%20heads%2C%20we%20define%20learnable%20linear%20projection%20parameter%20matrices%3A%0A%0A%20%20%20%20%24%24W_i%5EQ%20%5Cin%20%5Cmathbb%7BR%7D%5E%7Bd_%7B%5Ctext%7Bmodel%7D%7D%20%5Ctimes%20d_k%7D%2C%20%5Cquad%20W_i%5EK%20%5Cin%20%5Cmathbb%7BR%7D%5E%7Bd_%7B%5Ctext%7Bmodel%7D%7D%20%5Ctimes%20d_k%7D%2C%20%5Cquad%20W_i%5EV%20%5Cin%20%5Cmathbb%7BR%7D%5E%7Bd_%7B%5Ctext%7Bmodel%7D%7D%20%5Ctimes%20d_v%7D%24%24%0A%0A%20%20%20%20for%20head%20index%20%24i%20%5Cin%20%5C%7B1%2C%202%2C%20%5Cdots%2C%20h%5C%7D%24%2C%20where%20standard%20practice%20sets%3A%0A%0A%20%20%20%20%24%24d_k%20%3D%20d_v%20%3D%20%5Cfrac%7Bd_%7B%5Ctext%7Bmodel%7D%7D%7D%7Bh%7D%24%24%0A%0A%20%20%20%20For%20each%20head%20%24i%24%2C%20the%20scaled%20dot-product%20attention%20is%20computed%20in%20its%20designated%20subspace%3A%0A%0A%20%20%20%20%24%24%5Ctext%7Bhead%7D_i%20%3D%20%5Coperatorname%7BAttention%7D(X%20W_i%5EQ%2C%20X%20W_i%5EK%2C%20X%20W_i%5EV)%20%3D%20%5Coperatorname%7BSoftmax%7D%5Cleft(%5Cfrac%7BQ_i%20K_i%5E%5Ctop%7D%7B%5Csqrt%7Bd_k%7D%7D%5Cright)%20V_i%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BN%20%5Ctimes%20d_v%7D%24%24%0A%0A%20%20%20%20The%20outputs%20of%20all%20%24h%24%20heads%20are%20concatenated%20horizontally%20and%20projected%20by%20the%20final%20linear%20matrix%20%24W%5EO%20%5Cin%20%5Cmathbb%7BR%7D%5E%7B(h%20%5Ccdot%20d_v)%20%5Ctimes%20d_%7B%5Ctext%7Bmodel%7D%7D%7D%24%3A%0A%0A%20%20%20%20%24%24%5Coperatorname%7BMultiHead%7D(Q%2C%20K%2C%20V)%20%3D%20%5Coperatorname%7BConcat%7D(%5Ctext%7Bhead%7D_1%2C%20%5Ctext%7Bhead%7D_2%2C%20%5Cdots%2C%20%5Ctext%7Bhead%7D_h)%20W%5EO%24%24%0A%0A%20%20%20%20Since%20%24h%20%5Ccdot%20d_v%20%3D%20h%20%5Ccdot%20(d_%7B%5Ctext%7Bmodel%7D%7D%20%2F%20h)%20%3D%20d_%7B%5Ctext%7Bmodel%7D%7D%24%2C%20the%20output%20dimension%20matches%20the%20input%20dimension%20exactly%20(%24N%20%5Ctimes%20d_%7B%5Ctext%7Bmodel%7D%7D%24)%2C%20permitting%20clean%20residual%20addition%3A%0A%0A%20%20%20%20%24%24X_%7B%5Ctext%7Bout%7D%7D%20%3D%20%5Coperatorname%7BLayerNorm%7D(X%20%2B%20%5Coperatorname%7BMultiHead%7D(X))%24%24%0A%0A%20%20%20%20%23%23%23%202.%20Computational%20Equivalence%20and%20Unified%20Tensor%20Projections%0A%0A%20%20%20%20A%20naive%20implementation%20of%20MHA%20would%20execute%20%243h%24%20separate%20matrix%20multiplications%2C%20incurring%20unacceptable%20dispatch%20latency.%20In%20practice%2C%20all%20heads%20are%20fused%20into%20unified%20linear%20projections%3A%0A%0A%20%20%20%20%24%24W_Q%2C%20W_K%2C%20W_V%20%5Cin%20%5Cmathbb%7BR%7D%5E%7Bd_%7B%5Ctext%7Bmodel%7D%7D%20%5Ctimes%20d_%7B%5Ctext%7Bmodel%7D%7D%7D%24%24%0A%0A%20%20%20%20%24%24%5Ctilde%7BQ%7D%20%3D%20X%20W_Q%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BB%20%5Ctimes%20N%20%5Ctimes%20d_%7B%5Ctext%7Bmodel%7D%7D%7D%24%24%0A%0A%20%20%20%20The%20tensor%20is%20then%20reshaped%20and%20permuted%20across%20head%20and%20sequence%20dimensions%3A%0A%0A%20%20%20%20%24%24%5Ctilde%7BQ%7D%20%5Cxrightarrow%7B%5Ctext%7Breshape%7D%7D%20(B%2C%20N%2C%20h%2C%20d_k)%20%5Cxrightarrow%7B%5Ctext%7Bpermute%7D(0%2C%202%2C%201%2C%203)%7D%20(B%2C%20h%2C%20N%2C%20d_k)%24%24%0A%0A%20%20%20%20The%20attention%20affinity%20scores%20for%20all%20%24h%24%20heads%20across%20the%20entire%20batch%20%24B%24%20are%20computed%20simultaneously%20via%20batched%20matrix%20multiplication%3A%0A%0A%20%20%20%20%24%24S%20%3D%20%5Cfrac%7B%5Ctilde%7BQ%7D%20%5Ctilde%7BK%7D%5E%5Ctop%7D%7B%5Csqrt%7Bd_k%7D%7D%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BB%20%5Ctimes%20h%20%5Ctimes%20N%20%5Ctimes%20N%7D%24%24%0A%0A%20%20%20%20The%20total%20FLOP%20count%20for%20computing%20Multi-Head%20Attention%20across%20all%20%24h%24%20heads%20is%3A%0A%0A%20%20%20%20%24%24%5Cmathcal%7BO%7D%5Cleft(4%20N%20d_%7B%5Ctext%7Bmodel%7D%7D%5E2%20%2B%202%20N%5E2%20d_%7B%5Ctext%7Bmodel%7D%7D%5Cright)%24%24%0A%0A%20%20%20%20Remarkably%2C%20this%20computational%20cost%20is%20**identical**%20to%20single-head%20attention%20of%20dimension%20%24d_%7B%5Ctext%7Bmodel%7D%7D%24%2C%20demonstrating%20that%20MHA%20enhances%20representational%20capacity%20without%20increasing%20FLOP%20overhead.%0A%0A%20%20%20%20%23%23%23%203.%20Representation%20Rank%20and%20Subspace%20Orthogonality%0A%0A%20%20%20%20Each%20individual%20attention%20head%20produces%20an%20output%20matrix%20of%20maximum%20rank%3A%0A%0A%20%20%20%20%24%24%5Coperatorname%7Brank%7D(%5Ctext%7Bhead%7D_i)%20%5Cle%20%5Cmin(N%2C%20d_v)%20%3D%20%5Cmin%5Cleft(N%2C%20%5Cfrac%7Bd_%7B%5Ctext%7Bmodel%7D%7D%7D%7Bh%7D%5Cright)%24%24%0A%0A%20%20%20%20A%20single%20head%20with%20%24d_v%20%3C%20d_%7B%5Ctext%7Bmodel%7D%7D%24%20is%20strictly%20rank-constrained.%20By%20concatenating%20%24h%24%20independent%20heads%2C%20the%20rank%20of%20the%20concatenated%20matrix%20satisfies%3A%0A%0A%20%20%20%20%24%24%5Coperatorname%7Brank%7D%5Cleft(%5Coperatorname%7BConcat%7D(%5Ctext%7Bhead%7D_1%2C%20%5Cdots%2C%20%5Ctext%7Bhead%7D_h)%5Cright)%20%5Cle%20%5Cmin%5Cleft(N%2C%20%5Csum_%7Bi%3D1%7D%5Eh%20%5Coperatorname%7Brank%7D(%5Ctext%7Bhead%7D_i)%5Cright)%20%5Cle%20%5Cmin(N%2C%20d_%7B%5Ctext%7Bmodel%7D%7D)%24%24%0A%0A%20%20%20%20Multi-Head%20Attention%20enables%20the%20network%20to%20reconstruct%20full-rank%20transformations%20across%20the%20complete%20%24d_%7B%5Ctext%7Bmodel%7D%7D%24%20space%20while%20allowing%20each%20individual%20head%20to%20specialize%20in%20an%20isolated%20semantic%20subspace.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(go%2C%20make_subplots%2C%20mo%2C%20np)%3A%0A%20%20%20%20%23%20Sentence%20simulating%20diverse%20linguistic%20head%20specialization%0A%20%20%20%20_tokens%20%3D%20%5B%22The%22%2C%20%22astronomer%22%2C%20%22observed%22%2C%20%22the%22%2C%20%22distant%22%2C%20%22galaxy%22%2C%20%22carefully%22%5D%0A%20%20%20%20_n_tok%20%3D%20len(_tokens)%0A%0A%20%20%20%20%23%20Synthetic%20specialized%20attention%20weight%20patterns%20for%204%20heads%0A%20%20%20%20%23%20Head%201%3A%20Positional%20%2F%20Local%20Context%20(attends%20to%20immediately%20preceding%20token)%0A%20%20%20%20_A_head1%20%3D%20np.zeros((_n_tok%2C%20_n_tok))%0A%20%20%20%20for%20_i%20in%20range(_n_tok)%3A%0A%20%20%20%20%20%20%20%20for%20_j%20in%20range(_i%20%2B%201)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20_j%20%3D%3D%20_i%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_A_head1%5B_i%2C%20_j%5D%20%3D%200.6%0A%20%20%20%20%20%20%20%20%20%20%20%20elif%20_j%20%3D%3D%20_i%20-%201%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_A_head1%5B_i%2C%20_j%5D%20%3D%200.35%0A%20%20%20%20%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20_A_head1%5B_i%2C%20_j%5D%20%3D%200.05%20%2F%20max(1%2C%20_i%20-%201)%0A%20%20%20%20%20%20%20%20_A_head1%5B_i%5D%20%2F%3D%20np.sum(_A_head1%5B_i%5D)%0A%0A%20%20%20%20%23%20Head%202%3A%20Syntactic%20Subject-Verb-Object%20Dependency%20(%22observed%22%20attends%20to%20%22astronomer%22%20and%20%22galaxy%22)%0A%20%20%20%20_A_head2%20%3D%20np.zeros((_n_tok%2C%20_n_tok))%0A%20%20%20%20for%20_i%20in%20range(_n_tok)%3A%0A%20%20%20%20%20%20%20%20_A_head2%5B_i%2C%20_i%5D%20%3D%200.2%0A%20%20%20%20%20%20%20%20if%20_tokens%5B_i%5D%20%3D%3D%20%22observed%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_A_head2%5B_i%2C%20_tokens.index(%22astronomer%22)%5D%20%3D%200.45%0A%20%20%20%20%20%20%20%20%20%20%20%20_A_head2%5B_i%2C%20_tokens.index(%22galaxy%22)%5D%20%3D%200.35%0A%20%20%20%20%20%20%20%20elif%20_tokens%5B_i%5D%20%3D%3D%20%22carefully%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_A_head2%5B_i%2C%20_tokens.index(%22observed%22)%5D%20%3D%200.7%0A%20%20%20%20%20%20%20%20%20%20%20%20_A_head2%5B_i%2C%20_i%5D%20%3D%200.3%0A%20%20%20%20%20%20%20%20elif%20_tokens%5B_i%5D%20%3D%3D%20%22galaxy%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_A_head2%5B_i%2C%20_tokens.index(%22distant%22)%5D%20%3D%200.5%0A%20%20%20%20%20%20%20%20%20%20%20%20_A_head2%5B_i%2C%20_i%5D%20%3D%200.3%0A%20%20%20%20%20%20%20%20%20%20%20%20_A_head2%5B_i%2C%20_tokens.index(%22observed%22)%5D%20%3D%200.2%0A%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_A_head2%5B_i%2C%20%3A%20_i%20%2B%201%5D%20%3D%201.0%20%2F%20(_i%20%2B%201)%0A%20%20%20%20%20%20%20%20_A_head2%5B_i%5D%20%2F%3D%20np.sum(_A_head2%5B_i%5D)%0A%0A%20%20%20%20%23%20Head%203%3A%20Long-range%20Global%20Broadcast%20(High%20attention%20to%20initial%20root%20token%20%22The%22%20%2F%20%22astronomer%22)%0A%20%20%20%20_A_head3%20%3D%20np.zeros((_n_tok%2C%20_n_tok))%0A%20%20%20%20for%20_i%20in%20range(_n_tok)%3A%0A%20%20%20%20%20%20%20%20_A_head3%5B_i%2C%200%5D%20%3D%200.4%20%20%23%20%22The%22%0A%20%20%20%20%20%20%20%20_A_head3%5B_i%2C%201%5D%20%3D%200.4%20%20%23%20%22astronomer%22%0A%20%20%20%20%20%20%20%20_A_head3%5B_i%2C%20_i%5D%20%2B%3D%200.2%0A%20%20%20%20%20%20%20%20_A_head3%5B_i%2C%20%3A%20_i%20%2B%201%5D%20%2F%3D%20np.sum(_A_head3%5B_i%2C%20%3A%20_i%20%2B%201%5D)%0A%20%20%20%20%20%20%20%20_A_head3%5B_i%2C%20_i%20%2B%201%20%3A%5D%20%3D%200.0%0A%0A%20%20%20%20%23%20Head%204%3A%20Modifier%20%2F%20Adjective-Noun%20Coupling%20(%22distant%22%20-%3E%20%22galaxy%22%2C%20%22carefully%22%20-%3E%20%22observed%22)%0A%20%20%20%20_A_head4%20%3D%20np.zeros((_n_tok%2C%20_n_tok))%0A%20%20%20%20for%20_i%20in%20range(_n_tok)%3A%0A%20%20%20%20%20%20%20%20_A_head4%5B_i%2C%20_i%5D%20%3D%200.25%0A%20%20%20%20%20%20%20%20if%20_tokens%5B_i%5D%20%3D%3D%20%22distant%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_A_head4%5B_i%2C%20_tokens.index(%22galaxy%22)%5D%20%3D%200.65%0A%20%20%20%20%20%20%20%20elif%20_tokens%5B_i%5D%20%3D%3D%20%22galaxy%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_A_head4%5B_i%2C%20_tokens.index(%22distant%22)%5D%20%3D%200.65%0A%20%20%20%20%20%20%20%20elif%20_tokens%5B_i%5D%20%3D%3D%20%22carefully%22%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_A_head4%5B_i%2C%20_tokens.index(%22observed%22)%5D%20%3D%200.65%0A%20%20%20%20%20%20%20%20else%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_A_head4%5B_i%2C%20%3A%20_i%20%2B%201%5D%20%3D%201.0%20%2F%20(_i%20%2B%201)%0A%20%20%20%20%20%20%20%20_A_head4%5B_i%5D%20%2F%3D%20np.sum(_A_head4%5B_i%5D)%0A%0A%20%20%20%20fig%20%3D%20make_subplots(%0A%20%20%20%20%20%20%20%20rows%3D2%2C%0A%20%20%20%20%20%20%20%20cols%3D2%2C%0A%20%20%20%20%20%20%20%20subplot_titles%3D%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%3Cb%3EHead%201%3A%20Local%20%2F%20Positional%20Window%3C%2Fb%3E%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%3Cb%3EHead%202%3A%20Syntactic%20Dependency%20(Verb-Object)%3C%2Fb%3E%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%3Cb%3EHead%203%3A%20Global%20Root%20Broadcast%3C%2Fb%3E%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%3Cb%3EHead%204%3A%20Modifier-Noun%20Semantic%20Binding%3C%2Fb%3E%22%2C%0A%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%20%20%20%20horizontal_spacing%3D0.12%2C%0A%20%20%20%20%20%20%20%20vertical_spacing%3D0.18%2C%0A%20%20%20%20)%0A%0A%20%20%20%20_heads_data%20%3D%20%5B%0A%20%20%20%20%20%20%20%20(_A_head1%2C%201%2C%201%2C%20%22Purples%22)%2C%0A%20%20%20%20%20%20%20%20(_A_head2%2C%201%2C%202%2C%20%22Blues%22)%2C%0A%20%20%20%20%20%20%20%20(_A_head3%2C%202%2C%201%2C%20%22Teal%22)%2C%0A%20%20%20%20%20%20%20%20(_A_head4%2C%202%2C%202%2C%20%22Viridis%22)%2C%0A%20%20%20%20%5D%0A%0A%20%20%20%20for%20_mat%2C%20_r%2C%20_c%2C%20_cmap%20in%20_heads_data%3A%0A%20%20%20%20%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20%20%20%20%20go.Heatmap(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20z%3D_mat%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20x%3D_tokens%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20y%3D_tokens%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20colorscale%3D_cmap%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20zmin%3D0.0%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20zmax%3D0.8%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20showscale%3D(_r%20%3D%3D%201%20and%20_c%20%3D%3D%202)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20colorbar%3Ddict(title%3D%22Weight%22%2C%20x%3D1.02)%20if%20(_r%20%3D%3D%201%20and%20_c%20%3D%3D%202)%20else%20None%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20row%3D_r%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20col%3D_c%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20fig.update_xaxes(title_text%3D%22Key%20Token%22%2C%20row%3D_r%2C%20col%3D_c)%0A%20%20%20%20%20%20%20%20fig.update_yaxes(title_text%3D%22Query%20Token%22%2C%20autorange%3D%22reversed%22%2C%20row%3D_r%2C%20col%3D_c)%0A%0A%20%20%20%20fig.update_layout(%0A%20%20%20%20%20%20%20%20template%3D%22plotly_white%22%2C%0A%20%20%20%20%20%20%20%20height%3D620%2C%0A%20%20%20%20%20%20%20%20margin%3Ddict(l%3D40%2C%20r%3D40%2C%20t%3D70%2C%20b%3D50)%2C%0A%20%20%20%20)%0A%0A%20%20%20%20viz%20%3D%20mo.ui.plotly(fig)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(mo%2C%20nn%2C%20np%2C%20pd%2C%20torch)%3A%0A%20%20%20%20%23%20Vectorized%20NumPy%20Multi-Head%20Attention%20Implementation%0A%20%20%20%20def%20numpy_multi_head_attention(X%2C%20W_q%2C%20W_k%2C%20W_v%2C%20W_o%2C%20n_heads%3D4%2C%20mask%3DNone)%3A%0A%20%20%20%20%20%20%20%20%22%22%22Pure%20NumPy%20implementation%20of%20fused%20Multi-Head%20Attention.%0A%0A%20%20%20%20%20%20%20%20Args%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20X%3A%20Input%20tensor%20of%20shape%20(B%2C%20N%2C%20d_model)%0A%20%20%20%20%20%20%20%20%20%20%20%20W_q%2C%20W_k%2C%20W_v%3A%20Projection%20matrices%20of%20shape%20(d_model%2C%20d_model)%0A%20%20%20%20%20%20%20%20%20%20%20%20W_o%3A%20Output%20projection%20matrix%20of%20shape%20(d_model%2C%20d_model)%0A%20%20%20%20%20%20%20%20%20%20%20%20n_heads%3A%20Number%20of%20attention%20heads%0A%20%20%20%20%20%20%20%20%20%20%20%20mask%3A%20Optional%20boolean%20causal%20mask%20(N%2C%20N)%0A%20%20%20%20%20%20%20%20%22%22%22%0A%20%20%20%20%20%20%20%20B%2C%20N%2C%20d_model%20%3D%20X.shape%0A%20%20%20%20%20%20%20%20d_k%20%3D%20d_model%20%2F%2F%20n_heads%0A%0A%20%20%20%20%20%20%20%20%23%201.%20Unified%20linear%20projections%0A%20%20%20%20%20%20%20%20Q%20%3D%20np.dot(X%2C%20W_q).reshape(B%2C%20N%2C%20n_heads%2C%20d_k).transpose(0%2C%202%2C%201%2C%203)%20%20%23%20(B%2C%20h%2C%20N%2C%20d_k)%0A%20%20%20%20%20%20%20%20K%20%3D%20np.dot(X%2C%20W_k).reshape(B%2C%20N%2C%20n_heads%2C%20d_k).transpose(0%2C%202%2C%201%2C%203)%0A%20%20%20%20%20%20%20%20V%20%3D%20np.dot(X%2C%20W_v).reshape(B%2C%20N%2C%20n_heads%2C%20d_k).transpose(0%2C%202%2C%201%2C%203)%0A%0A%20%20%20%20%20%20%20%20%23%202.%20Scaled%20Dot-Product%20Attention%20across%20all%20heads%20in%20parallel%0A%20%20%20%20%20%20%20%20scores%20%3D%20np.matmul(Q%2C%20K.transpose(0%2C%201%2C%203%2C%202))%20%2F%20np.sqrt(d_k)%20%20%23%20(B%2C%20h%2C%20N%2C%20N)%0A%0A%20%20%20%20%20%20%20%20if%20mask%20is%20not%20None%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20scores%20%3D%20np.where(mask%2C%20scores%2C%20-1e9)%0A%0A%20%20%20%20%20%20%20%20exp_s%20%3D%20np.exp(scores%20-%20np.max(scores%2C%20axis%3D-1%2C%20keepdims%3DTrue))%0A%20%20%20%20%20%20%20%20weights%20%3D%20exp_s%20%2F%20np.sum(exp_s%2C%20axis%3D-1%2C%20keepdims%3DTrue)%0A%0A%20%20%20%20%20%20%20%20%23%203.%20Value%20aggregation%20and%20head%20concatenation%0A%20%20%20%20%20%20%20%20head_outputs%20%3D%20np.matmul(weights%2C%20V)%20%20%23%20(B%2C%20h%2C%20N%2C%20d_k)%0A%20%20%20%20%20%20%20%20head_outputs%20%3D%20head_outputs.transpose(0%2C%202%2C%201%2C%203).reshape(B%2C%20N%2C%20d_model)%20%20%23%20(B%2C%20N%2C%20d_model)%0A%0A%20%20%20%20%20%20%20%20%23%204.%20Final%20linear%20output%20projection%0A%20%20%20%20%20%20%20%20output%20%3D%20np.dot(head_outputs%2C%20W_o)%0A%20%20%20%20%20%20%20%20return%20output%2C%20weights%0A%0A%20%20%20%20%23%20Verification%201%3A%20Shape%20and%20Dimension%20Conservation%0A%20%20%20%20np.random.seed(42)%0A%20%20%20%20_B%2C%20_N%2C%20_dm%2C%20_h%20%3D%202%2C%206%2C%2032%2C%204%0A%20%20%20%20_X%20%3D%20np.random.randn(_B%2C%20_N%2C%20_dm)%0A%0A%20%20%20%20_Wq%20%3D%20np.random.randn(_dm%2C%20_dm)%20%2F%20np.sqrt(_dm)%0A%20%20%20%20_Wk%20%3D%20np.random.randn(_dm%2C%20_dm)%20%2F%20np.sqrt(_dm)%0A%20%20%20%20_Wv%20%3D%20np.random.randn(_dm%2C%20_dm)%20%2F%20np.sqrt(_dm)%0A%20%20%20%20_Wo%20%3D%20np.random.randn(_dm%2C%20_dm)%20%2F%20np.sqrt(_dm)%0A%0A%20%20%20%20_out_np%2C%20_w_np%20%3D%20numpy_multi_head_attention(_X%2C%20_Wq%2C%20_Wk%2C%20_Wv%2C%20_Wo%2C%20n_heads%3D_h)%0A%0A%20%20%20%20df_shape_verif%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Layer_Component%22%3A%20%22Input%20Sequence%20X%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Tensor_Shape%22%3A%20f%22%7B_X.shape%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Expected_Dimension%22%3A%20f%22(%7B_B%7D%2C%20%7B_N%7D%2C%20%7B_dm%7D)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Status%22%3A%20%22Match%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Layer_Component%22%3A%20%22Multi-Head%20Attention%20Weights%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Tensor_Shape%22%3A%20f%22%7B_w_np.shape%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Expected_Dimension%22%3A%20f%22(%7B_B%7D%2C%20%7B_h%7D%2C%20%7B_N%7D%2C%20%7B_N%7D)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Status%22%3A%20%22Match%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Layer_Component%22%3A%20%22Contextualized%20Output%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Tensor_Shape%22%3A%20f%22%7B_out_np.shape%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Expected_Dimension%22%3A%20f%22(%7B_B%7D%2C%20%7B_N%7D%2C%20%7B_dm%7D)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Status%22%3A%20%22Preserved%20(Ready%20for%20Residual)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%0A%20%20%20%20%23%20Verification%202%3A%20PyTorch%20Production%20MultiHeadAttention%20Module%0A%20%20%20%20class%20PyTorchMHA(nn.Module)%3A%0A%20%20%20%20%20%20%20%20def%20__init__(self%2C%20d_model%3D32%2C%20n_heads%3D4)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20super().__init__()%0A%20%20%20%20%20%20%20%20%20%20%20%20self.d_model%20%3D%20d_model%0A%20%20%20%20%20%20%20%20%20%20%20%20self.n_heads%20%3D%20n_heads%0A%20%20%20%20%20%20%20%20%20%20%20%20self.head_dim%20%3D%20d_model%20%2F%2F%20n_heads%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20self.q_proj%20%3D%20nn.Linear(d_model%2C%20d_model)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.k_proj%20%3D%20nn.Linear(d_model%2C%20d_model)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.v_proj%20%3D%20nn.Linear(d_model%2C%20d_model)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.out_proj%20%3D%20nn.Linear(d_model%2C%20d_model)%0A%0A%20%20%20%20%20%20%20%20def%20forward(self%2C%20x%2C%20is_causal%3DFalse)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20B%2C%20N%2C%20C%20%3D%20x.shape%0A%20%20%20%20%20%20%20%20%20%20%20%20q%20%3D%20self.q_proj(x).view(B%2C%20N%2C%20self.n_heads%2C%20self.head_dim).transpose(1%2C%202)%0A%20%20%20%20%20%20%20%20%20%20%20%20k%20%3D%20self.k_proj(x).view(B%2C%20N%2C%20self.n_heads%2C%20self.head_dim).transpose(1%2C%202)%0A%20%20%20%20%20%20%20%20%20%20%20%20v%20%3D%20self.v_proj(x).view(B%2C%20N%2C%20self.n_heads%2C%20self.head_dim).transpose(1%2C%202)%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20scores%20%3D%20(q%20%40%20k.transpose(-2%2C%20-1))%20%2F%20(self.head_dim**0.5)%0A%20%20%20%20%20%20%20%20%20%20%20%20if%20is_causal%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20mask%20%3D%20torch.triu(torch.full((N%2C%20N)%2C%20float(%22-inf%22)%2C%20device%3Dx.device)%2C%20diagonal%3D1)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20scores%20%3D%20scores%20%2B%20mask%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20weights%20%3D%20torch.softmax(scores%2C%20dim%3D-1)%0A%20%20%20%20%20%20%20%20%20%20%20%20ctx%20%3D%20(weights%20%40%20v).transpose(1%2C%202).contiguous().view(B%2C%20N%2C%20C)%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20self.out_proj(ctx)%2C%20weights%0A%0A%20%20%20%20torch.manual_seed(42)%0A%20%20%20%20mha_torch%20%3D%20PyTorchMHA(d_model%3D32%2C%20n_heads%3D4)%0A%20%20%20%20x_tensor%20%3D%20torch.randn(2%2C%206%2C%2032)%0A%20%20%20%20with%20torch.no_grad()%3A%0A%20%20%20%20%20%20%20%20out_pt%2C%20weights_pt%20%3D%20mha_torch(x_tensor%2C%20is_causal%3DTrue)%0A%0A%20%20%20%20%23%20Verification%203%3A%20Inter-Head%20Subspace%20Orthogonality%20Audit%0A%20%20%20%20%23%20Measure%20cosine%20similarity%20between%20projection%20subspaces%20of%20different%20heads%0A%20%20%20%20W_q_heads%20%3D%20mha_torch.q_proj.weight.detach().numpy().reshape(4%2C%208%2C%2032)%20%20%23%20(h%2C%20d_k%2C%20d_model)%0A%20%20%20%20subspace_corr%20%3D%20np.zeros((4%2C%204))%0A%20%20%20%20for%20_i%20in%20range(4)%3A%0A%20%20%20%20%20%20%20%20for%20_j%20in%20range(4)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20Compute%20matrix%20cosine%20similarity%3A%20Tr(A%20B%5ET)%20%2F%20(%7C%7CA%7C%7C_F%20%7C%7CB%7C%7C_F)%0A%20%20%20%20%20%20%20%20%20%20%20%20frob_i%20%3D%20np.linalg.norm(W_q_heads%5B_i%5D%2C%20%22fro%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20frob_j%20%3D%20np.linalg.norm(W_q_heads%5B_j%5D%2C%20%22fro%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20inner_prod%20%3D%20np.trace(np.dot(W_q_heads%5B_i%5D%2C%20W_q_heads%5B_j%5D.T))%0A%20%20%20%20%20%20%20%20%20%20%20%20subspace_corr%5B_i%2C%20_j%5D%20%3D%20inner_prod%20%2F%20(frob_i%20*%20frob_j)%0A%0A%20%20%20%20df_orthogonality%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Head_Pair%22%3A%20f%22Head%20%7B_i%2B1%7D%20vs%20Head%20%7B_j%2B1%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Frobenius_Cosine_Correlation%22%3A%20f%22%7Bsubspace_corr%5B_i%2C%20_j%5D%3A.4f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Subspace_Overlap%22%3A%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Self-Identity%20(1.0000)%22%20if%20_i%20%3D%3D%20_j%20else%20(%22Orthogonal%20Subspaces%22%20if%20abs(subspace_corr%5B_i%2C%20_j%5D)%20%3C%200.25%20else%20%22Moderate%20Correlation%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20_i%20in%20range(4)%0A%20%20%20%20%20%20%20%20%20%20%20%20for%20_j%20in%20range(_i%2C%204)%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%0A%20%20%20%20table_shapes%20%3D%20mo.ui.table(df_shape_verif)%0A%20%20%20%20table_ortho%20%3D%20mo.ui.table(df_orthogonality)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
32e0bd5eaed160ae5fbdd14e0ed36dd7