{"metadata":{"kernelspec":{"name":"ir","display_name":"R","language":"R"},"language_info":{"name":"R","codemirror_mode":"r","pygments_lexer":"r","mimetype":"text/x-r-source","file_extension":".r","version":"4.4.0"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":118765,"databundleVersionId":15231210,"sourceType":"competition"}],"dockerImageVersionId":30749,"isInternetEnabled":true,"language":"r","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Load the data","metadata":{}},{"cell_type":"code","source":"train_labels <- read.table(\n  \"/kaggle/input/stanford-rna-3d-folding-2/train_labels.csv\",\n  header = TRUE,\n  sep = \",\"\n)","metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:21:18.593127Z","iopub.execute_input":"2026-02-03T08:21:18.594801Z","iopub.status.idle":"2026-02-03T08:21:48.969839Z","shell.execute_reply":"2026-02-03T08:21:48.968487Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Lets plot basic stio from train_labels.csv","metadata":{}},{"cell_type":"code","source":"head(train_labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:21:48.971744Z","iopub.execute_input":"2026-02-03T08:21:48.973107Z","iopub.status.idle":"2026-02-03T08:21:48.994557Z","shell.execute_reply":"2026-02-03T08:21:48.992811Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"summary(x[, c(\"x_1\", \"y_1\", \"z_1\")])\n\n# counts\ncat(\"Number of rows:\", nrow(x), \"\\n\")\ncat(\"Unique residues:\", length(unique(x$resname)), \"\\n\")\ncat(\"Chains:\", unique(x$chain), \"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:21:48.997254Z","iopub.execute_input":"2026-02-03T08:21:48.998191Z","iopub.status.idle":"2026-02-03T08:21:49.104047Z","shell.execute_reply":"2026-02-03T08:21:49.023713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"install.packages(\"ggplot2\")\nlibrary(\"ggplot2\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:23:02.854255Z","iopub.execute_input":"2026-02-03T08:23:02.855495Z","iopub.status.idle":"2026-02-03T08:23:50.781276Z","shell.execute_reply":"2026-02-03T08:23:50.780008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ggplot(data = train_labels,\n    mapping = aes(x = x_1)) +\n    geom_density() + \n    theme_classic()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:23:50.783116Z","iopub.execute_input":"2026-02-03T08:23:50.784228Z","iopub.status.idle":"2026-02-03T08:23:55.495403Z","shell.execute_reply":"2026-02-03T08:23:55.494105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ggplot(data = train_labels,\n    mapping = aes(x = y_1)) +\n    geom_density() + \n    theme_classic()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:23:55.497157Z","iopub.execute_input":"2026-02-03T08:23:55.498227Z","iopub.status.idle":"2026-02-03T08:23:58.823021Z","shell.execute_reply":"2026-02-03T08:23:58.821724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ggplot(data = train_labels,\n       mapping = aes(x = z_1)) +\n    geom_density() + \n    theme_classic()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:23:58.82493Z","iopub.execute_input":"2026-02-03T08:23:58.825942Z","iopub.status.idle":"2026-02-03T08:24:02.362846Z","shell.execute_reply":"2026-02-03T08:24:02.360842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ggplot(data = train_labels,\n       mapping = aes(x = resname)) +\n    geom_bar() + \n    theme_classic()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:24:02.365783Z","iopub.execute_input":"2026-02-03T08:24:02.367044Z","iopub.status.idle":"2026-02-03T08:24:06.329174Z","shell.execute_reply":"2026-02-03T08:24:06.327849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"prop.table(table(x = train_labels$resname))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:24:06.331084Z","iopub.execute_input":"2026-02-03T08:24:06.332113Z","iopub.status.idle":"2026-02-03T08:24:06.746582Z","shell.execute_reply":"2026-02-03T08:24:06.745415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"install.packages(\"BiocManager\")\nBiocManager::install(\"Biostrings\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:24:06.748311Z","iopub.execute_input":"2026-02-03T08:24:06.749249Z","iopub.status.idle":"2026-02-03T08:26:53.849158Z","shell.execute_reply":"2026-02-03T08:26:53.847789Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### plot one sample","metadata":{}},{"cell_type":"code","source":"library(Biostrings)\n\nmsa <- readAAStringSet(\"/kaggle/input/stanford-rna-3d-folding-2/MSA/157D.MSA.fasta\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:26:53.851137Z","iopub.execute_input":"2026-02-03T08:26:53.85221Z","iopub.status.idle":"2026-02-03T08:26:55.324898Z","shell.execute_reply":"2026-02-03T08:26:55.323669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"msa\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:26:55.326671Z","iopub.execute_input":"2026-02-03T08:26:55.327854Z","iopub.status.idle":"2026-02-03T08:26:55.570171Z","shell.execute_reply":"2026-02-03T08:26:55.56879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BiocManager::install(\"bio3d\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:26:55.572124Z","iopub.execute_input":"2026-02-03T08:26:55.573201Z","iopub.status.idle":"2026-02-03T08:27:41.732858Z","shell.execute_reply":"2026-02-03T08:27:41.731685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"library(bio3d)\nlibrary(scatterplot3d)\n\ncif_file <- \"/kaggle/input/stanford-rna-3d-folding-2/PDB_RNA/100d.cif\"\nstructure <- read.cif(cif_file)\n\nrna_atoms <- structure$atom[structure$atom$type == \"ATOM\", ]\n\nx <- rna_atoms$x\ny <- rna_atoms$y\nz <- rna_atoms$z\n\nhead(rna_atoms)\n\nscatterplot3d(\n  x, y, z,\n  color = \"blue\",\n  pch = 16,\n  main = \"3D RNA Structure: 100d\"\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:27:41.734674Z","iopub.execute_input":"2026-02-03T08:27:41.735622Z","iopub.status.idle":"2026-02-03T08:27:42.000506Z","shell.execute_reply":"2026-02-03T08:27:41.998367Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### each prediction colored differently","metadata":{}},{"cell_type":"code","source":"atom_colors <- as.numeric(as.factor(rna_atoms$eleno))\n\n# 2. Plot with the dynamic color vector\nscatterplot3d(\n  rna_atoms$x, rna_atoms$y, rna_atoms$z,\n  color = atom_colors,  # Colors based on atom type\n  pch = 16,\n  main = \"3D RNA Structure: 100d (Colored by Element)\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:27:42.00369Z","iopub.execute_input":"2026-02-03T08:27:42.005331Z","iopub.status.idle":"2026-02-03T08:27:42.093374Z","shell.execute_reply":"2026-02-03T08:27:42.091481Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### train_sequences.csv","metadata":{}},{"cell_type":"code","source":"train_sequences <- read.csv(\"/kaggle/input/stanford-rna-3d-folding-2/train_sequences.csv\", \n                  header = TRUE, stringsAsFactors = FALSE)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:27:42.096276Z","iopub.execute_input":"2026-02-03T08:27:42.097998Z","iopub.status.idle":"2026-02-03T08:27:42.830921Z","shell.execute_reply":"2026-02-03T08:27:42.829464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"head(train_sequences)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:27:42.834012Z","iopub.execute_input":"2026-02-03T08:27:42.835162Z","iopub.status.idle":"2026-02-03T08:27:42.858761Z","shell.execute_reply":"2026-02-03T08:27:42.85671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"names(train_sequences)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:27:42.861734Z","iopub.execute_input":"2026-02-03T08:27:42.862819Z","iopub.status.idle":"2026-02-03T08:27:42.8759Z","shell.execute_reply":"2026-02-03T08:27:42.874685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences$seq_length <- nchar(train_sequences$sequence)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:27:42.877813Z","iopub.execute_input":"2026-02-03T08:27:42.878819Z","iopub.status.idle":"2026-02-03T08:27:42.89836Z","shell.execute_reply":"2026-02-03T08:27:42.896924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ggplot(train_sequences, aes(x = seq_length)) +\n  geom_density(fill = \"purple\", alpha = 0.5) +\n  labs(title = \"Density of Sequence Lengths\", x = \"Sequence Length\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:27:42.900259Z","iopub.execute_input":"2026-02-03T08:27:42.901241Z","iopub.status.idle":"2026-02-03T08:27:43.103733Z","shell.execute_reply":"2026-02-03T08:27:43.102449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"library(dplyr)\n\ndf <- train_labels %>%\n  mutate(target_id = sub(\"_.*$\", \"\", ID)) %>%\n  inner_join(\n    train_sequences %>% select(target_id, seq_length),\n    by = \"target_id\"\n  )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:27:43.105554Z","iopub.execute_input":"2026-02-03T08:27:43.106524Z","iopub.status.idle":"2026-02-03T08:27:46.556513Z","shell.execute_reply":"2026-02-03T08:27:46.554939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cor(df$x_1, df$seq_length, use = \"complete.obs\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:27:46.558409Z","iopub.execute_input":"2026-02-03T08:27:46.559406Z","iopub.status.idle":"2026-02-03T08:27:47.150541Z","shell.execute_reply":"2026-02-03T08:27:47.149044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fit <- lm(x_1 ~ seq_length, data = df)\nsummary(fit)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:27:47.15275Z","iopub.execute_input":"2026-02-03T08:27:47.153986Z","iopub.status.idle":"2026-02-03T08:27:53.30212Z","shell.execute_reply":"2026-02-03T08:27:53.300894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot(df$seq_length, df$x_1,\n     xlab = \"Sequence length\",\n     ylab = \"x_1\",\n     main = \"x_1 vs Sequence Length\",\n     pch = 16, cex = 0.5)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:27:53.304044Z","iopub.execute_input":"2026-02-03T08:27:53.305041Z","iopub.status.idle":"2026-02-03T08:28:51.649003Z","shell.execute_reply":"2026-02-03T08:28:51.647661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# For seq_length\nQ1_seq <- quantile(df$seq_length, 0.25)\nQ3_seq <- quantile(df$seq_length, 0.75)\nIQR_seq <- Q3_seq - Q1_seq\n\n\ndf_both <- subset(df, \n                   seq_length >= (Q1_seq - 1.5 * IQR_seq) & seq_length <= (Q3_seq + 1.5 * IQR_seq))\n\n# 3. Plot the cleaned data\nplot(df_both$seq_length, df_both$x_1,\n     xlab = \"Sequence length\",\n     ylab = \"x_1\",\n     main = \"x_1 vs Sequence Length (Outliers Removed)\",\n     pch = 16, \n     cex = 0.5,\n     col = adjustcolor(\"darkblue\", alpha.f = 0.5)) # Added transparency to see density","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:28:51.650936Z","iopub.execute_input":"2026-02-03T08:28:51.651912Z","iopub.status.idle":"2026-02-03T08:29:52.269413Z","shell.execute_reply":"2026-02-03T08:29:52.267469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# For seq_length\nQ1_seq <- quantile(df$seq_length, 0.25)\nQ3_seq <- quantile(df$seq_length, 0.75)\nIQR_seq <- Q3_seq - Q1_seq\n\n\ndf_both <- subset(df, \n                   seq_length >= (Q1_seq - 1.5 * IQR_seq) & seq_length <= (Q3_seq + 1.5 * IQR_seq))\n\n# 3. Plot the cleaned data\nplot(df_both$seq_length, df_both$y_1,\n     xlab = \"Sequence length\",\n     ylab = \"y_1\",\n     main = \"y_1 vs Sequence Length (Outliers Removed)\",\n     pch = 16, \n     cex = 0.5,\n     col = adjustcolor(\"darkblue\", alpha.f = 0.5)) # Added transparency to see density","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:29:52.27253Z","iopub.execute_input":"2026-02-03T08:29:52.273971Z","iopub.status.idle":"2026-02-03T08:30:52.456167Z","shell.execute_reply":"2026-02-03T08:30:52.453437Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Z vs seq len + pol reg","metadata":{}},{"cell_type":"code","source":"Q1_seq <- quantile(df$seq_length, 0.25)\nQ3_seq <- quantile(df$seq_length, 0.75)\nIQR_seq <- Q3_seq - Q1_seq\n\ndf_both <- subset(df, \n                  seq_length >= (Q1_seq - 1.5 * IQR_seq) & \n                  seq_length <= (Q3_seq + 1.5 * IQR_seq))\n\n\npoly_model <- lm(z_1 ~ poly(seq_length, 10), data = df_both)\n\nx_range <- seq(min(df_both$seq_length), max(df_both$seq_length), length.out = 100)\ny_preds <- predict(poly_model, newdata = data.frame(seq_length = x_range))\n\nplot(df_both$seq_length, df_both$z_1,\n     xlab = \"Sequence length\",\n     ylab = \"z_1\",\n     main = \"z_1 vs Sequence Length with Polynomial Fit\",\n     pch = 16, \n     cex = 0.5,\n     col = adjustcolor(\"darkblue\", alpha.f = 0.5))\n\nlines(x_range, y_preds, col = \"red\", lwd = 2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T09:20:26.663851Z","iopub.execute_input":"2026-02-03T09:20:26.665213Z","iopub.status.idle":"2026-02-03T09:21:39.577782Z","shell.execute_reply":"2026-02-03T09:21:39.576256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cor(df_both$seq_length, df_both$z_1, method = \"pearson\", use = \"complete.obs\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:31:52.374205Z","iopub.execute_input":"2026-02-03T08:31:52.375229Z","iopub.status.idle":"2026-02-03T08:31:52.55146Z","shell.execute_reply":"2026-02-03T08:31:52.550192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cor(df_both$seq_length, df_both$x_1, method = \"pearson\", use = \"complete.obs\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:31:52.553282Z","iopub.execute_input":"2026-02-03T08:31:52.554249Z","iopub.status.idle":"2026-02-03T08:31:52.734356Z","shell.execute_reply":"2026-02-03T08:31:52.732927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cor(df_both$seq_length, df_both$y_1, method = \"pearson\", use = \"complete.obs\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:31:52.73637Z","iopub.execute_input":"2026-02-03T08:31:52.737482Z","iopub.status.idle":"2026-02-03T08:31:52.916089Z","shell.execute_reply":"2026-02-03T08:31:52.914888Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Relation between x,y,z","metadata":{}},{"cell_type":"markdown","source":"### x and z","metadata":{}},{"cell_type":"code","source":"plot(df$x_1, df$z_1,\n     xlab = \"x_1\",\n     ylab = \"z_1\",\n     main = \"z_1 vs x_1\",\n     pch = 16, \n     cex = 0.5,\n     col = adjustcolor(\"darkblue\", alpha.f = 0.5))\n\n# linear regression line\nabline(lm(z_1 ~ x_1, data = df), col = \"red\", lwd = 2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:31:52.917909Z","iopub.execute_input":"2026-02-03T08:31:52.918907Z","iopub.status.idle":"2026-02-03T08:32:56.243495Z","shell.execute_reply":"2026-02-03T08:32:56.242006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_clean <- df[is.finite(df$x_1) & is.finite(df$z_1), ]\n\nlibrary(MASS)\ndens <- kde2d(df_clean$x_1, df_clean$z_1, n = 50)\n\nfilled.contour(dens, \n               color.palette = function(n) hcl.colors(n, \"Viridis\"),\n               main = \"Density Distribution (Cleaned Data)\",\n               xlab = \"x_1\",\n               ylab = \"z_1\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:32:56.245487Z","iopub.execute_input":"2026-02-03T08:32:56.246568Z","iopub.status.idle":"2026-02-03T08:33:36.069944Z","shell.execute_reply":"2026-02-03T08:33:36.068559Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### x and y","metadata":{}},{"cell_type":"code","source":"\nplot(df$x_1, df$y_1,\n     xlab = \"x_1\",\n     ylab = \"y_1\",\n     main = \"y_1 vs x_1\",\n     pch = 16, \n     cex = 0.5,\n     col = adjustcolor(\"darkblue\", alpha.f = 0.5)) # Added transparency to see density\nabline(lm(x_1 ~ y_1, data = df), col = \"red\", lwd = 2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:33:36.071852Z","iopub.execute_input":"2026-02-03T08:33:36.07286Z","iopub.status.idle":"2026-02-03T08:34:39.83236Z","shell.execute_reply":"2026-02-03T08:34:39.830885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_clean <- df[is.finite(df$x_1) & is.finite(df$y_1), ]\n\nlibrary(MASS)\ndens <- kde2d(df_clean$x_1, df_clean$y_1, n = 50)\n\nfilled.contour(dens, \n               color.palette = function(n) hcl.colors(n, \"Viridis\"),\n               main = \"Density Distribution (Cleaned Data)\",\n               xlab = \"x_1\",\n               ylab = \"y_1\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:34:39.834322Z","iopub.execute_input":"2026-02-03T08:34:39.835401Z","iopub.status.idle":"2026-02-03T08:35:11.463436Z","shell.execute_reply":"2026-02-03T08:35:11.461826Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### z and y","metadata":{}},{"cell_type":"code","source":"\nplot(df$z_1, df$y_1,\n     xlab = \"z_1\",\n     ylab = \"y_1\",\n     main = \"y_1 vs z_1\",\n     pch = 16, \n     cex = 0.5,\n     col = adjustcolor(\"darkblue\", alpha.f = 0.5)) # Added transparency to see density\nabline(lm(z_1 ~ y_1, data = df), col = \"red\", lwd = 2)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:35:11.465598Z","iopub.execute_input":"2026-02-03T08:35:11.466711Z","iopub.status.idle":"2026-02-03T08:36:15.145257Z","shell.execute_reply":"2026-02-03T08:36:15.143701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_clean <- df[is.finite(df$y_1) & is.finite(df$z_1), ]\n\nlibrary(MASS)\ndens <- kde2d(df_clean$y_1, df_clean$z_1, n = 50)\n\nfilled.contour(dens, \n               color.palette = function(n) hcl.colors(n, \"Viridis\"),\n               main = \"Density Distribution (Cleaned Data)\",\n               xlab = \"y_1\",\n               ylab = \"z_1\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:36:15.147186Z","iopub.execute_input":"2026-02-03T08:36:15.148293Z","iopub.status.idle":"2026-02-03T08:36:47.607811Z","shell.execute_reply":"2026-02-03T08:36:47.606091Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3D ","metadata":{}},{"cell_type":"code","source":"res <- persp(dens, \n             phi = 30, theta = 30,  # Angles to rotate the view\n             shade = 0.5,           # Adds a shadow for depth\n             col = \"lightblue\", \n             border = NA,           # Removes the grid lines for a smooth look\n             main = \"3D Density Distribution\",\n             xlab = \"x_1\", ylab = \"z_1\", zlab = \"Density\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:36:47.611042Z","iopub.execute_input":"2026-02-03T08:36:47.612757Z","iopub.status.idle":"2026-02-03T08:36:47.717212Z","shell.execute_reply":"2026-02-03T08:36:47.715702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"library(plotly)\n\n# Plotly uses the z-matrix from the density object\nplot_ly(x = dens$x, y = dens$y, z = dens$z) %>% \n  add_surface() %>%\n  layout(\n    title = \"Interactive 3D Density Map\",\n    scene = list(\n      xaxis = list(title = \"x_1\"),\n      yaxis = list(title = \"z_1\"),\n      zaxis = list(title = \"Density\")\n    )\n  )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:36:47.719206Z","iopub.execute_input":"2026-02-03T08:36:47.720235Z","iopub.status.idle":"2026-02-03T08:36:48.580442Z","shell.execute_reply":"2026-02-03T08:36:48.579051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ggplot(missing_by_structure, aes(x = missing_pct)) +\n  geom_histogram(binwidth = 5, fill = \"steelblue\", color = \"white\") +\n  labs(\n    title = \"Distribution of Missing Coordinate Percentage by Structure\",\n    x = \"Percentage of Missing Coordinates\",\n    y = \"Number of Structures\"\n  ) +\n  theme_classic()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-03T08:36:48.58216Z","iopub.execute_input":"2026-02-03T08:36:48.583192Z","iopub.status.idle":"2026-02-03T08:36:48.597217Z","shell.execute_reply":"2026-02-03T08:36:48.592547Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dup_ids <- train_labels %>%\n  group_by(ID) %>%\n  filter(n() > 1)\n\ncat(\"Duplicate IDs in train_labels:\", nrow(dup_ids), \"\\n\")\nif(nrow(dup_ids) > 0) {\n  cat(\"Examples of duplicates:\\n\")\n  print(head(dup_ids, 10))\n}\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ggplot(missing_by_structure, aes(x = total_residues, y = missing_pct)) +\n  geom_point(alpha = 0.3, color = \"darkred\") +\n  labs(\n    title = \"Missingness vs Structure Size\",\n    x = \"Total Residues in Structure\",\n    y = \"% Missing Coordinates\"\n  ) +\n  theme_classic()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Gaussian Processes","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}