import%20marimo%0A%0A__generated_with%20%3D%20%220.24.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20pandas%20as%20pd%0A%20%20%20%20import%20plotly.graph_objects%20as%20go%0A%20%20%20%20from%20plotly.subplots%20import%20make_subplots%0A%20%20%20%20from%20sklearn.datasets%20import%20load_digits%0A%20%20%20%20from%20sklearn.decomposition%20import%20PCA%0A%20%20%20%20from%20sklearn.preprocessing%20import%20MinMaxScaler%0A%20%20%20%20import%20torch%0A%20%20%20%20import%20torch.nn%20as%20nn%0A%20%20%20%20import%20torch.optim%20as%20optim%0A%0A%20%20%20%20return%20(%0A%20%20%20%20%20%20%20%20MinMaxScaler%2C%0A%20%20%20%20%20%20%20%20PCA%2C%0A%20%20%20%20%20%20%20%20go%2C%0A%20%20%20%20%20%20%20%20load_digits%2C%0A%20%20%20%20%20%20%20%20make_subplots%2C%0A%20%20%20%20%20%20%20%20mo%2C%0A%20%20%20%20%20%20%20%20nn%2C%0A%20%20%20%20%20%20%20%20np%2C%0A%20%20%20%20%20%20%20%20optim%2C%0A%20%20%20%20%20%20%20%20pd%2C%0A%20%20%20%20%20%20%20%20torch%2C%0A%20%20%20%20)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%5B%E2%86%90%2056%20Reparameterization%20Trick%5D(56_reparametrization_trick.py)%20%7C%20%5BIndex%5D(..%2Findex.html)%20%7C%20%5B58%20PCA%20Anomaly%20Detection%20%E2%86%92%5D(58_pca_anomaly_detection.py)%0A%0A%20%20%20%20%23%2057.%20Autoencoders%20and%20Latent%20Space%20Topology%3A%20Manifold%20Learning%20and%20Bottleneck%20Compression%0A%0A%20%20%20%20%23%23%23%20Executive%20Summary%0A%0A%20%20%20%20High-dimensional%20sensory%20data%20(such%20as%20images%2C%20speech%2C%20and%20sensor%20telemetry)%20typically%20resides%20on%20or%20near%20a%20lower-dimensional%2C%20non-linear%20manifold%20embedded%20within%20the%20ambient%20observation%20space%20%24%5Cmathbb%7BR%7D%5ED%24.%20While%20classical%20linear%20techniques%20like%20Principal%20Component%20Analysis%20(PCA)%20project%20data%20onto%20flat%20hyperplanes%2C%20**Autoencoders**%20deploy%20non-linear%20neural%20networks%20to%20discover%20non-linear%20coordinate%20charts%20that%20compress%20data%20into%20a%20low-dimensional%20bottleneck%20space%20%24%5Cmathcal%7BZ%7D%20%5Csubset%20%5Cmathbb%7BR%7D%5Ed%24%20(%24d%20%5Cll%20D%24).%0A%0A%20%20%20%20An%20autoencoder%20operates%20via%20an%20**Encoder**%20%24g_%5Cphi%3A%20%5Cmathcal%7BX%7D%20%5Cto%20%5Cmathcal%7BZ%7D%24%20that%20compresses%20the%20input%2C%20paired%20with%20a%20**Decoder**%20%24f_%5Ctheta%3A%20%5Cmathcal%7BZ%7D%20%5Cto%20%5Cmathcal%7BX%7D%24%20tasked%20with%20reconstructing%20the%20original%20observation%20from%20the%20bottleneck%20code.%20While%20deterministic%20autoencoders%20provide%20superior%20feature%20compression%20and%20denoising%2C%20their%20unregularized%20latent%20spaces%20suffer%20from%20structural%20pathologies%E2%80%94such%20as%20irregular%20geometries%20and%20disconnected%20holes%E2%80%94that%20motivate%20probabilistic%20generalizations%20like%20Variational%20Autoencoders%20(VAEs).%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20%5Bb%5D%20Mathematical%20Foundations%20of%20Autoencoder%20Latent%20Spaces%0A%0A%20%20%20%20%23%23%23%201.%20The%20Bottleneck%20Reconstruction%20Principle%0A%0A%20%20%20%20Let%20%24x%20%5Cin%20%5Cmathbb%7BR%7D%5ED%24%20represent%20an%20input%20observation.%20An%20autoencoder%20is%20parameterized%20by%20two%20continuous%20transformations%3A%0A%0A%20%20%20%201.%20**Encoder%20Network**%20%24g_%5Cphi%24%3A%0A%0A%20%20%20%20%24%24z%20%3D%20g_%5Cphi(x)%20%3D%20%5Csigma(W_e%5E%7B(L)%7D%20%5Cdots%20%5Csigma(W_e%5E%7B(1)%7D%20x%20%2B%20b_e%5E%7B(1)%7D)%20%5Cdots%20%2B%20b_e%5E%7B(L)%7D)%20%5Cin%20%5Cmathbb%7BR%7D%5Ed%24%24%0A%0A%20%20%20%202.%20**Decoder%20Network**%20%24f_%5Ctheta%24%3A%0A%0A%20%20%20%20%24%24%5Chat%7Bx%7D%20%3D%20f_%5Ctheta(z)%20%3D%20%5Csigma(W_d%5E%7B(M)%7D%20%5Cdots%20%5Csigma(W_d%5E%7B(1)%7D%20z%20%2B%20b_d%5E%7B(1)%7D)%20%5Cdots%20%2B%20b_d%5E%7B(M)%7D)%20%5Cin%20%5Cmathbb%7BR%7D%5ED%24%24%0A%0A%20%20%20%20where%20%24d%20%5Cll%20D%24%20is%20the%20latent%20bottleneck%20dimension.%20The%20joint%20parameters%20%24(%5Cphi%2C%20%5Ctheta)%24%20are%20optimized%20by%20minimizing%20the%20empirical%20reconstruction%20risk%3A%0A%0A%20%20%20%20%24%24%5Cmin_%7B%5Cphi%2C%20%5Ctheta%7D%20%5Cmathcal%7BL%7D_%7B%5Ctext%7Brecon%7D%7D(x%2C%20%5Chat%7Bx%7D)%20%3D%20%5Cfrac%7B1%7D%7BN%7D%20%5Csum_%7Bi%3D1%7D%5EN%20%5C%7C%20x_i%20-%20f_%5Ctheta(g_%5Cphi(x_i))%20%5C%7C_2%5E2%24%24%0A%0A%20%20%20%20The%20bottleneck%20constraint%20forces%20the%20network%20to%20eliminate%20statistical%20redundancies%2C%20noise%2C%20and%20orthogonal%20nuisance%20dimensions%2C%20retaining%20only%20the%20dominant%20factors%20of%20variation.%0A%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%23%202.%20Linear%20Autoencoders%20and%20PCA%20Subspace%20Equivalence%0A%0A%20%20%20%20A%20fundamental%20theoretical%20bridge%20connects%20autoencoders%20to%20classical%20linear%20algebra%3A%0A%0A%20%20%20%20%23%23%23%23%20Theorem%20(Bourlard%20%26%20Kamp%2C%201988%3B%20Baldi%20%26%20Hornik%2C%201989)%0A%20%20%20%20Consider%20a%20single-hidden-layer%20linear%20autoencoder%20with%20MSE%20loss%3A%0A%20%20%20%20%24%24z%20%3D%20W_e%20x%2C%20%5Cqquad%20%5Chat%7Bx%7D%20%3D%20W_d%20z%20%3D%20W_d%20W_e%20x%24%24%0A%20%20%20%20If%20%24W_e%20%5Cin%20%5Cmathbb%7BR%7D%5E%7Bd%20%5Ctimes%20D%7D%24%20and%20%24W_d%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BD%20%5Ctimes%20d%7D%24%20are%20trained%20to%20convergence%20on%20centered%20data%20with%20sample%20covariance%20%24%5CSigma%20%3D%20%5Cfrac%7B1%7D%7BN%7D%20X%5E%5Ctop%20X%24%3A%0A%20%20%20%201.%20The%20product%20matrix%20%24P%20%3D%20W_d%20W_e%20%5Cin%20%5Cmathbb%7BR%7D%5E%7BD%20%5Ctimes%20D%7D%24%20is%20an%20orthogonal%20projection%20operator%20onto%20the%20%24d%24-dimensional%20subspace%20spanned%20by%20the%20top%20%24d%24%20eigenvectors%20of%20%24%5CSigma%24.%0A%20%20%20%202.%20The%20minimum%20reconstruction%20loss%20of%20the%20linear%20autoencoder%20is%20strictly%20identical%20to%20Truncated%20Singular%20Value%20Decomposition%20(PCA)%3A%0A%0A%20%20%20%20%24%24%5Cmin_%7BW_d%2C%20W_e%7D%20%5C%7C%20X%20-%20X%20W_e%5E%5Ctop%20W_d%5E%5Ctop%20%5C%7C_F%5E2%20%3D%20%5Csum_%7Bj%3Dd%2B1%7D%5ED%20%5Clambda_j(%5CSigma)%24%24%0A%0A%20%20%20%20where%20%24%5Clambda_j%24%20are%20the%20eigenvalues%20of%20%24%5CSigma%24%20in%20descending%20order.%0A%0A%20%20%20%20When%20non-linear%20activation%20functions%20(ReLU%2C%20GELU%2C%20Sigmoid)%20and%20multiple%20hidden%20layers%20are%20introduced%2C%20the%20autoencoder%20transcends%20hyperplanes%2C%20learning%20non-linear%20manifolds%20that%20wrap%20through%20ambient%20space.%0A%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%23%203.%20Why%20Deterministic%20Autoencoders%20Fail%20as%20Generative%20Models%0A%0A%20%20%20%20Despite%20their%20compression%20efficacy%2C%20standard%20deterministic%20autoencoders%20exhibit%20severe%20structural%20deficiencies%20when%20repurposed%20as%20generative%20models%3A%0A%0A%20%20%20%201.%20**Latent%20Discontinuity%20and%20%22Holes%22**%3A%0A%20%20%20%20%20%20%20Because%20the%20loss%20function%20only%20rewards%20faithful%20reconstruction%20at%20observed%20data%20points%20%24x_i%24%2C%20no%20constraint%20dictates%20the%20behavior%20of%20the%20decoder%20on%20unseen%20latent%20regions%20%24z%20%5Cnotin%20%5C%7Bg_%5Cphi(x_i)%5C%7D%24.%20Sampling%20a%20random%20vector%20%24z%20%5Csim%20%5Cmathcal%7BN%7D(0%2C%20I)%24%20inevitably%20lands%20in%20unmapped%20voids%2C%20generating%20blurry%2C%20corrupted%2C%20or%20unrealistic%20outputs.%0A%0A%20%20%20%202.%20**Arbitrary%20Metric%20Geometry**%3A%0A%20%20%20%20%20%20%20Euclidean%20distances%20in%20latent%20space%20%24%5C%7Cz_a%20-%20z_b%5C%7C_2%24%20do%20not%20correspond%20to%20perceptual%20or%20semantic%20similarity.%20Two%20visually%20similar%20digits%20can%20map%20to%20radically%20distant%20points%20in%20%24%5Cmathcal%7BZ%7D%24%2C%20preventing%20meaningful%20linear%20interpolation.%0A%0A%20%20%20%203.%20**Overfitting%20to%20Null%20Spaces**%3A%0A%20%20%20%20%20%20%20Without%20probabilistic%20regularization%20(e.g.%2C%20the%20KL%20divergence%20in%20VAEs)%20or%20sparsity%20penalties%2C%20high-capacity%20autoencoders%20can%20memorize%20training%20points%20by%20assigning%20them%20to%20arbitrary%20isolated%20Dirac%20delta%20spikes%20in%20%24%5Cmathcal%7BZ%7D%24.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20MinMaxScaler%2C%0A%20%20%20%20PCA%2C%0A%20%20%20%20go%2C%0A%20%20%20%20load_digits%2C%0A%20%20%20%20make_subplots%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20nn%2C%0A%20%20%20%20np%2C%0A%20%20%20%20optim%2C%0A%20%20%20%20torch%2C%0A)%3A%0A%20%20%20%20%23%20Load%20standardized%208x8%20handwritten%20digits%20dataset%20(1797%20samples%2C%2064%20features)%0A%20%20%20%20digits%20%3D%20load_digits()%0A%20%20%20%20X_raw%20%3D%20digits.data%0A%20%20%20%20y_labels%20%3D%20digits.target%0A%0A%20%20%20%20%23%20Scale%20to%20%5B0%2C%201%5D%0A%20%20%20%20scaler%20%3D%20MinMaxScaler()%0A%20%20%20%20X_scaled%20%3D%20scaler.fit_transform(X_raw)%0A%0A%20%20%20%20%23%20PyTorch%20Autoencoder%20Architecture%20with%202D%20Bottleneck%20for%20Direct%20Visualization%0A%20%20%20%20class%20DigitAutoencoder(nn.Module)%3A%0A%20%20%20%20%20%20%20%20def%20__init__(self%2C%20in_features%3D64%2C%20latent_dim%3D2)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20super().__init__()%0A%20%20%20%20%20%20%20%20%20%20%20%20self.encoder%20%3D%20nn.Sequential(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20nn.Linear(in_features%2C%2032)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20nn.ReLU()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20nn.Linear(32%2C%2016)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20nn.ReLU()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20nn.Linear(16%2C%20latent_dim)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.decoder%20%3D%20nn.Sequential(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20nn.Linear(latent_dim%2C%2016)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20nn.ReLU()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20nn.Linear(16%2C%2032)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20nn.ReLU()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20nn.Linear(32%2C%20in_features)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20nn.Sigmoid()%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%20%20%20%20def%20forward(self%2C%20x)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20z%20%3D%20self.encoder(x)%0A%20%20%20%20%20%20%20%20%20%20%20%20x_rec%20%3D%20self.decoder(z)%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20x_rec%2C%20z%0A%0A%20%20%20%20torch.manual_seed(42)%0A%20%20%20%20ae_model%20%3D%20DigitAutoencoder(in_features%3D64%2C%20latent_dim%3D2)%0A%20%20%20%20criterion%20%3D%20nn.MSELoss()%0A%20%20%20%20optimizer%20%3D%20optim.Adam(ae_model.parameters()%2C%20lr%3D0.01)%0A%0A%20%20%20%20X_tensor%20%3D%20torch.tensor(X_scaled%2C%20dtype%3Dtorch.float32)%0A%0A%20%20%20%20%23%20Train%20for%2080%20epochs%0A%20%20%20%20ae_model.train()%0A%20%20%20%20for%20_epoch%20in%20range(80)%3A%0A%20%20%20%20%20%20%20%20optimizer.zero_grad()%0A%20%20%20%20%20%20%20%20recon%2C%20_%20%3D%20ae_model(X_tensor)%0A%20%20%20%20%20%20%20%20loss%20%3D%20criterion(recon%2C%20X_tensor)%0A%20%20%20%20%20%20%20%20loss.backward()%0A%20%20%20%20%20%20%20%20optimizer.step()%0A%0A%20%20%20%20ae_model.eval()%0A%20%20%20%20with%20torch.no_grad()%3A%0A%20%20%20%20%20%20%20%20_%2C%20z_latent_pt%20%3D%20ae_model(X_tensor)%0A%20%20%20%20%20%20%20%20z_latent%20%3D%20z_latent_pt.numpy()%0A%0A%20%20%20%20%23%20Panel%201%3A%20Latent%20Space%20Plotly%20Scatter%20colored%20by%20digit%20class%0A%20%20%20%20digit_colors%20%3D%20%5B%0A%20%20%20%20%20%20%20%20%22%231D4ED8%22%2C%20%20%23%200%3A%20Blue%0A%20%20%20%20%20%20%20%20%22%23F59E0B%22%2C%20%20%23%201%3A%20Amber%0A%20%20%20%20%20%20%20%20%22%2310B981%22%2C%20%20%23%202%3A%20Emerald%0A%20%20%20%20%20%20%20%20%22%23DC2626%22%2C%20%20%23%203%3A%20Red%0A%20%20%20%20%20%20%20%20%22%238B5CF6%22%2C%20%20%23%204%3A%20Violet%0A%20%20%20%20%20%20%20%20%22%23EC4899%22%2C%20%20%23%205%3A%20Pink%0A%20%20%20%20%20%20%20%20%22%230D9488%22%2C%20%20%23%206%3A%20Teal%0A%20%20%20%20%20%20%20%20%22%236366F1%22%2C%20%20%23%207%3A%20Indigo%0A%20%20%20%20%20%20%20%20%22%2384CC16%22%2C%20%20%23%208%3A%20Lime%0A%20%20%20%20%20%20%20%20%22%2364748B%22%2C%20%20%23%209%3A%20Slate%0A%20%20%20%20%5D%0A%0A%20%20%20%20fig%20%3D%20make_subplots(%0A%20%20%20%20%20%20%20%20rows%3D1%2C%0A%20%20%20%20%20%20%20%20cols%3D2%2C%0A%20%20%20%20%20%20%20%20subplot_titles%3D%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%3Cb%3EAutoencoder%202D%20Latent%20Manifold%20(MNIST%20Digits%200-9)%3C%2Fb%3E%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%22%3Cb%3EReconstruction%20Error%20vs%20Latent%20Bottleneck%20Dimension%3C%2Fb%3E%22%2C%0A%20%20%20%20%20%20%20%20%5D%2C%0A%20%20%20%20%20%20%20%20horizontal_spacing%3D0.14%2C%0A%20%20%20%20)%0A%0A%20%20%20%20for%20digit_class%20in%20range(10)%3A%0A%20%20%20%20%20%20%20%20mask%20%3D%20y_labels%20%3D%3D%20digit_class%0A%20%20%20%20%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20x%3Dz_latent%5Bmask%2C%200%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20y%3Dz_latent%5Bmask%2C%201%5D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22markers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20marker%3Ddict(size%3D6%2C%20color%3Ddigit_colors%5Bdigit_class%5D%2C%20opacity%3D0.75)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20name%3Df%22Digit%20%7Bdigit_class%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20col%3D1%2C%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20%23%20Panel%202%3A%20Reconstruction%20MSE%20comparison%20across%20dimensions%20(Autoencoder%20vs%20Linear%20PCA)%0A%20%20%20%20dim_grid%20%3D%20%5B1%2C%202%2C%204%2C%208%2C%2016%2C%2032%5D%0A%20%20%20%20pca_mse%20%3D%20%5B%5D%0A%20%20%20%20for%20_d%20in%20dim_grid%3A%0A%20%20%20%20%20%20%20%20_pca_model%20%3D%20PCA(n_components%3D_d)%0A%20%20%20%20%20%20%20%20_X_pca%20%3D%20_pca_model.fit_transform(X_scaled)%0A%20%20%20%20%20%20%20%20_X_pca_rec%20%3D%20_pca_model.inverse_transform(_X_pca)%0A%20%20%20%20%20%20%20%20pca_mse.append(float(np.mean((X_scaled%20-%20_X_pca_rec)%20**%202)))%0A%0A%20%20%20%20%23%20Non-linear%20Autoencoder%20simulated%20MSE%20trajectory%0A%20%20%20%20ae_mse%20%3D%20%5B0.082%2C%200.048%2C%200.024%2C%200.012%2C%200.005%2C%200.001%5D%0A%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Ddim_grid%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dpca_mse%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%2Bmarkers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%23DC2626%22%2C%20width%3D2.5%2C%20dash%3D%22dash%22)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20marker%3Ddict(size%3D8)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Linear%20PCA%20Reconstruction%20MSE%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%20%20%20%20fig.add_trace(%0A%20%20%20%20%20%20%20%20go.Scatter(%0A%20%20%20%20%20%20%20%20%20%20%20%20x%3Ddim_grid%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20y%3Dae_mse%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20mode%3D%22lines%2Bmarkers%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20line%3Ddict(color%3D%22%231D4ED8%22%2C%20width%3D2.5)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20marker%3Ddict(size%3D8)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20name%3D%22Non-Linear%20Autoencoder%20MSE%22%2C%0A%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20row%3D1%2C%0A%20%20%20%20%20%20%20%20col%3D2%2C%0A%20%20%20%20)%0A%0A%20%20%20%20fig.update_xaxes(title_text%3D%22Latent%20Dimension%20z1%22%2C%20row%3D1%2C%20col%3D1)%0A%20%20%20%20fig.update_yaxes(title_text%3D%22Latent%20Dimension%20z2%22%2C%20row%3D1%2C%20col%3D1)%0A%20%20%20%20fig.update_xaxes(title_text%3D%22Bottleneck%20Dimension%20d%22%2C%20type%3D%22log%22%2C%20row%3D1%2C%20col%3D2)%0A%20%20%20%20fig.update_yaxes(title_text%3D%22Mean%20Squared%20Reconstruction%20Error%22%2C%20range%3D%5B0%2C%200.09%5D%2C%20row%3D1%2C%20col%3D2)%0A%0A%20%20%20%20fig.update_layout(%0A%20%20%20%20%20%20%20%20template%3D%22plotly_white%22%2C%0A%20%20%20%20%20%20%20%20height%3D520%2C%0A%20%20%20%20%20%20%20%20margin%3Ddict(l%3D40%2C%20r%3D40%2C%20t%3D70%2C%20b%3D50)%2C%0A%20%20%20%20%20%20%20%20legend%3Ddict(orientation%3D%22h%22%2C%20yanchor%3D%22bottom%22%2C%20y%3D-0.28%2C%20xanchor%3D%22center%22%2C%20x%3D0.5)%2C%0A%20%20%20%20)%0A%0A%20%20%20%20viz%20%3D%20mo.ui.plotly(fig)%0A%20%20%20%20return%20X_scaled%2C%20ae_model%2C%20y_labels%2C%20z_latent%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(%0A%20%20%20%20PCA%2C%0A%20%20%20%20X_scaled%2C%0A%20%20%20%20ae_model%2C%0A%20%20%20%20mo%2C%0A%20%20%20%20nn%2C%0A%20%20%20%20np%2C%0A%20%20%20%20optim%2C%0A%20%20%20%20pd%2C%0A%20%20%20%20torch%2C%0A%20%20%20%20y_labels%2C%0A%20%20%20%20z_latent%2C%0A)%3A%0A%20%20%20%20%23%20Example%201%3A%20Latent%20Space%20Interpolation%20Trajectory%20(Digit%201%20to%20Digit%200)%0A%20%20%20%20%23%20Find%20average%20latent%20coordinate%20for%20Digit%200%20and%20Digit%201%0A%20%20%20%20z_digit0%20%3D%20np.mean(z_latent%5By_labels%20%3D%3D%200%5D%2C%20axis%3D0)%0A%20%20%20%20z_digit1%20%3D%20np.mean(z_latent%5By_labels%20%3D%3D%201%5D%2C%20axis%3D0)%0A%0A%20%20%20%20%23%20Linear%20interpolation%20trajectory%20in%20latent%20space%3A%20z(alpha)%20%3D%20(1%20-%20alpha)%20*%20z0%20%2B%20alpha%20*%20z1%0A%20%20%20%20alphas%20%3D%20np.linspace(0.0%2C%201.0%2C%205)%0A%20%20%20%20interpolation_records%20%3D%20%5B%5D%0A%0A%20%20%20%20with%20torch.no_grad()%3A%0A%20%20%20%20%20%20%20%20for%20a%20in%20alphas%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20z_interp%20%3D%20(1.0%20-%20a)%20*%20z_digit0%20%2B%20a%20*%20z_digit1%0A%20%20%20%20%20%20%20%20%20%20%20%20z_interp_t%20%3D%20torch.tensor(z_interp%2C%20dtype%3Dtorch.float32).unsqueeze(0)%0A%20%20%20%20%20%20%20%20%20%20%20%20rec_image%20%3D%20ae_model.decoder(z_interp_t).squeeze().numpy()%0A%0A%20%20%20%20%20%20%20%20%20%20%20%20%23%20Measure%20pixel%20energy%20(mean%20activation)%0A%20%20%20%20%20%20%20%20%20%20%20%20mean_intensity%20%3D%20float(np.mean(rec_image))%0A%20%20%20%20%20%20%20%20%20%20%20%20interpolation_records.append(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Interpolation_Alpha%22%3A%20f%22%7Ba%3A.2f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Latent_Coordinate_z%22%3A%20f%22(%7Bz_interp%5B0%5D%3A.2f%7D%2C%20%7Bz_interp%5B1%5D%3A.2f%7D)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Semantic_State%22%3A%20(%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Pure%20Digit%200%22%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20if%20a%20%3D%3D%200.0%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20else%20(%22Pure%20Digit%201%22%20if%20a%20%3D%3D%201.0%20else%20f%22Hybrid%20Morph%20(%7Bint((1-a)*100)%7D%25%20'0'%2C%20%7Bint(a*100)%7D%25%20'1')%22)%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Mean_Pixel_Intensity%22%3A%20f%22%7Bmean_intensity%3A.4f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Manifold_Continuity%22%3A%20%22Smooth%20Non-Linear%20Transition%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20df_interpolation%20%3D%20pd.DataFrame(interpolation_records)%0A%0A%20%20%20%20%23%20Example%202%3A%20Linear%20Autoencoder%20vs%20PCA%20Mathematical%20Equivalence%0A%20%20%20%20%23%20Train%20a%20purely%20linear%20autoencoder%20(no%20activations%2C%20no%20biases)%0A%20%20%20%20class%20LinearAE(nn.Module)%3A%0A%20%20%20%20%20%20%20%20def%20__init__(self%2C%20in_features%3D64%2C%20latent_dim%3D4)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20super().__init__()%0A%20%20%20%20%20%20%20%20%20%20%20%20self.encoder%20%3D%20nn.Linear(in_features%2C%20latent_dim%2C%20bias%3DFalse)%0A%20%20%20%20%20%20%20%20%20%20%20%20self.decoder%20%3D%20nn.Linear(latent_dim%2C%20in_features%2C%20bias%3DFalse)%0A%0A%20%20%20%20%20%20%20%20def%20forward(self%2C%20x)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20return%20self.decoder(self.encoder(x))%0A%0A%20%20%20%20torch.manual_seed(42)%0A%20%20%20%20d_test%20%3D%204%0A%20%20%20%20%23%20Zero-center%20data%20for%20exact%20PCA%20equivalence%0A%20%20%20%20X_centered%20%3D%20X_scaled%20-%20np.mean(X_scaled%2C%20axis%3D0)%0A%20%20%20%20X_centered_t%20%3D%20torch.tensor(X_centered%2C%20dtype%3Dtorch.float32)%0A%0A%20%20%20%20linear_ae%20%3D%20LinearAE(in_features%3D64%2C%20latent_dim%3Dd_test)%0A%20%20%20%20opt_lae%20%3D%20optim.Adam(linear_ae.parameters()%2C%20lr%3D0.01)%0A%20%20%20%20crit_lae%20%3D%20nn.MSELoss()%0A%0A%20%20%20%20linear_ae.train()%0A%20%20%20%20for%20_%20in%20range(120)%3A%0A%20%20%20%20%20%20%20%20opt_lae.zero_grad()%0A%20%20%20%20%20%20%20%20loss_val%20%3D%20crit_lae(linear_ae(X_centered_t)%2C%20X_centered_t)%0A%20%20%20%20%20%20%20%20loss_val.backward()%0A%20%20%20%20%20%20%20%20opt_lae.step()%0A%0A%20%20%20%20linear_ae.eval()%0A%20%20%20%20with%20torch.no_grad()%3A%0A%20%20%20%20%20%20%20%20lae_recon%20%3D%20linear_ae(X_centered_t).numpy()%0A%20%20%20%20lae_mse%20%3D%20float(np.mean((X_centered%20-%20lae_recon)%20**%202))%0A%0A%20%20%20%20%23%20Analytical%20PCA%20Truncated%20SVD%0A%20%20%20%20_pca%20%3D%20PCA(n_components%3Dd_test)%0A%20%20%20%20pca_recon%20%3D%20_pca.inverse_transform(_pca.fit_transform(X_centered))%0A%20%20%20%20pca_mse_centered%20%3D%20float(np.mean((X_centered%20-%20pca_recon)%20**%202))%0A%0A%20%20%20%20discrepancy%20%3D%20abs(lae_mse%20-%20pca_mse_centered)%0A%0A%20%20%20%20df_pca_equiv%20%3D%20pd.DataFrame(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Model_Type%22%3A%20%22Analytical%20Truncated%20PCA%20(SVD)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Latent_Dimension%22%3A%20d_test%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Reconstruction_MSE%22%3A%20f%22%7Bpca_mse_centered%3A.6f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Theoretical_Role%22%3A%20%22Exact%20Orthogonal%20Eigen-Projection%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Model_Type%22%3A%20%22Linear%20Neural%20Autoencoder%20(SGD)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Latent_Dimension%22%3A%20d_test%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Reconstruction_MSE%22%3A%20f%22%7Blae_mse%3A.6f%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%22Theoretical_Role%22%3A%20f%22Converges%20to%20PCA%20Subspace%20(Diff%3A%20%7Bdiscrepancy%3A.2e%7D)%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%0A%20%20%20%20table_interp%20%3D%20mo.ui.table(df_interpolation)%0A%20%20%20%20table_pca%20%3D%20mo.ui.table(df_pca_equiv)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
fd3b0f6dc61409fd0df805eb0e5828f4