{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":118765,"databundleVersionId":15231210,"sourceType":"competition"}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"execution":{"iopub.status.busy":"2026-02-07T19:08:57.26756Z","iopub.execute_input":"2026-02-07T19:08:57.268272Z","iopub.status.idle":"2026-02-07T19:09:12.876682Z","shell.execute_reply.started":"2026-02-07T19:08:57.268237Z","shell.execute_reply":"2026-02-07T19:09:12.875729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nimport pytorch_lightning as pl\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n# !pip install Bio\nfrom Bio import AlignIO","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T19:55:55.630611Z","iopub.execute_input":"2026-02-07T19:55:55.631646Z","iopub.status.idle":"2026-02-07T19:55:55.635855Z","shell.execute_reply.started":"2026-02-07T19:55:55.631611Z","shell.execute_reply":"2026-02-07T19:55:55.635271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RNAProcessor:\n    def __init__(self, config):\n        self.config = config\n        self.label_encoder = LabelEncoder()\n        \n    def prepare_data(self):\n        print(\">>> Loading Data...\")\n        seq_df = pd.read_csv(os.path.join(self.config['data_dir'], 'train_sequences.csv'))\n        meta_df = pd.read_csv(os.path.join(self.config['data_dir'], 'extra/rna_metadata.csv'))\n        seq_df['rna_type'] = 'Unknown' \n        # types = ['Ribozyme', 'tRNA', 'mRNA', 'Riboswitch', 'lincRNA']\n        # seq_df['rna_type'] = np.random.choice(types, size=len(seq_df))\n        seq_df = seq_df[['sequence', 'rna_type']].dropna()\n        self.classes = seq_df['rna_type'].unique()\n        seq_df['label'] = self.label_encoder.fit_transform(seq_df['rna_type'])\n        \n        return seq_df\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T19:55:58.011686Z","iopub.execute_input":"2026-02-07T19:55:58.011974Z","iopub.status.idle":"2026-02-07T19:55:58.017917Z","shell.execute_reply.started":"2026-02-07T19:55:58.011946Z","shell.execute_reply":"2026-02-07T19:55:58.017126Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RNADataset(Dataset):\n    def __init__(self, df, max_len):\n        self.sequences = df['sequence'].values\n        self.labels = df['label'].values\n        self.max_len = max_len\n        self.mapping = {'A': 1, 'C': 2, 'G': 3, 'U': 4, 'N': 0}\n\n    def __len__(self):\n        return len(self.sequences)\n\n    def __getitem__(self, idx):\n        seq = self.sequences[idx]\n        encoded = [self.mapping.get(s, 0) for s in seq[:self.max_len]]\n        padded = encoded + [0] * (self.max_len - len(encoded))\n        return torch.tensor(padded, dtype=torch.long), torch.tensor(self.labels[idx], dtype=torch.long)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T19:56:01.02975Z","iopub.execute_input":"2026-02-07T19:56:01.030486Z","iopub.status.idle":"2026-02-07T19:56:01.036135Z","shell.execute_reply.started":"2026-02-07T19:56:01.030454Z","shell.execute_reply":"2026-02-07T19:56:01.035593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RNADataModule(pl.LightningDataModule):\n    def __init__(self, df, config):\n        super().__init__()\n        self.df = df\n        self.config = config\n\n    def setup(self, stage=None):\n        train_df, val_df = train_test_split(self.df, test_size=0.2)\n        self.train_ds = RNADataset(train_df, self.config['max_len'])\n        self.val_ds = RNADataset(val_df, self.config['max_len'])\n\n    def train_dataloader(self):\n        return DataLoader(self.train_ds, batch_size=self.config['batch_size'], shuffle=True, num_workers=self.config['num_workers'], pin_memory=True)\n\n    def val_dataloader(self):\n        return DataLoader(self.val_ds, batch_size=self.config['batch_size'], num_workers=self.config['num_workers'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T19:56:04.127687Z","iopub.execute_input":"2026-02-07T19:56:04.128282Z","iopub.status.idle":"2026-02-07T19:56:04.133511Z","shell.execute_reply.started":"2026-02-07T19:56:04.128241Z","shell.execute_reply":"2026-02-07T19:56:04.13283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RNADataModule(pl.LightningDataModule):\n    def __init__(self, df, config):\n        super().__init__()\n        self.df = df\n        self.config = config\n\n    def setup(self, stage=None):\n        train_df, val_df = train_test_split(self.df, test_size=0.2)\n        self.train_ds = RNADataset(train_df, self.config['max_len'])\n        self.val_ds = RNADataset(val_df, self.config['max_len'])\n\n    def train_dataloader(self):\n        return DataLoader(self.train_ds, batch_size=self.config['batch_size'], \n                          shuffle=True, num_workers=self.config['num_workers'], pin_memory=True)\n\n    def val_dataloader(self):\n        return DataLoader(self.val_ds, batch_size=self.config['batch_size'], \n                          num_workers=self.config['num_workers'])","metadata":{"trusted":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RNAClassifier(pl.LightningModule):\n    def __init__(self, num_classes, config):\n        super().__init__()\n        self.save_hyperparameters()\n        self.embedding = nn.Embedding(5, config['embed_dim'], padding_idx=0)\n        self.lstm = nn.LSTM(config['embed_dim'], config['hidden_dim'], batch_first=True, bidirectional=True, num_layers=2, dropout=0.2)\n        self.attn_weight = nn.Linear(config['hidden_dim'] * 2, 1)\n        self.fc = nn.Linear(config['hidden_dim'] * 2, num_classes)\n        self.loss_fn = nn.CrossEntropyLoss()\n\n    def forward(self, x):\n        embed = self.embedding(x)\n        lstm_out, _ = self.lstm(embed)\n        attn_scores = self.attn_weight(lstm_out)\n        attn_weights = F.softmax(attn_scores, dim=1)\n        context = torch.sum(attn_weights * lstm_out, dim=1)\n        logits = self.fc(context)\n        return logits\n\n    def training_step(self, batch, batch_idx):\n        x, y = batch\n        logits = self(x)\n        loss = self.loss_fn(logits, y)\n        self.log('train_loss', loss, prog_bar=True)\n        return loss\n\n    def validation_step(self, batch, batch_idx):\n        x, y = batch\n        logits = self(x)\n        loss = self.loss_fn(logits, y)\n        preds = torch.argmax(logits, dim=1)\n        acc = (preds == y).float().mean()\n        self.log('val_loss', loss, prog_bar=True)\n        self.log('val_acc', acc, prog_bar=True)\n        return loss\n\n    def configure_optimizers(self):\n        return torch.optim.AdamW(self.parameters(), lr=self.hparams.config['lr'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T19:56:07.228532Z","iopub.execute_input":"2026-02-07T19:56:07.229186Z","iopub.status.idle":"2026-02-07T19:56:07.237003Z","shell.execute_reply.started":"2026-02-07T19:56:07.229158Z","shell.execute_reply":"2026-02-07T19:56:07.236372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if __name__ == '__main__':\n    CONFIG = {\n        'data_dir': '/kaggle/input/stanford-rna-3d-folding-2',\n        'max_len': 256,\n        'batch_size': 32,\n        'embed_dim': 128,\n        'hidden_dim': 256,\n        'lr': 1e-3,\n        'epochs': 10,\n        'num_workers': 4\n    }    \n    processor = RNAProcessor(CONFIG)\n\n    try:\n        df = processor.prepare_data()\n        num_classes = len(processor.classes)\n        print(f\"Found {num_classes} RNA classes: {processor.classes}\")\n        dm = RNADataModule(df, CONFIG)\n        model = RNAClassifier(num_classes=num_classes, config=CONFIG)\n        trainer = pl.Trainer(\n            max_epochs=CONFIG['epochs'],\n            accelerator='auto',\n            devices=1,\n            precision='16-mixed'\n        )\n\n        print(\">>> Starting Training...\")\n        trainer.fit(model, dm)\n        print(\">>> Training Complete!\")\n    except FileNotFoundError as e:\n        print(f\"Error: Could not find files. Please ensure the Kaggle dataset is attached.\\n{e}\")\n    except Exception as e:\n        print(f\"An unexpected error occurred: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T19:56:19.819845Z","iopub.execute_input":"2026-02-07T19:56:19.820416Z","iopub.status.idle":"2026-02-07T19:58:15.88206Z","shell.execute_reply.started":"2026-02-07T19:56:19.820385Z","shell.execute_reply":"2026-02-07T19:58:15.881261Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_rmsd(pred_coords, true_coords):\n    p_centered = pred_coords - pred_coords.mean(axis=0)\n    t_centered = true_coords - true_coords.mean(axis=0)\n\n    H = np.dot(p_centered.T, t_centered)\n    U, S, Vt = np.linalg.svd(H)\n    R = np.dot(Vt.T, U.T)\n\n    if np.linalg.det(R) < 0:\n        Vt[2, :] *= -1\n        R = np.dot(Vt.T, U.T)\n\n    p_rotated = np.dot(p_centered, R)\n    diff = p_rotated - t_centered\n    rmsd = np.sqrt((diff ** 2).sum() / len(p_rotated))\n    return rmsd\n\n\ndef validate_model(model, val_loader, device='cuda'):\n    model.eval()\n    scores = []\n    print(f\"Starting validation on {len(val_loader)} batches...\")\n    with torch.no_grad():\n        for batch_idx, (sequences, true_coords) in enumerate(val_loader):\n            inputs = sequences.to(device)\n            predictions = model(inputs) \n            for i in range(len(inputs)):\n                sample_scores = []\n                truth = true_coords[i].cpu().numpy()\n                for k in range(5):\n                    pred = predictions[i, k].cpu().numpy()\n                    mask = ~np.isnan(truth).any(axis=1)\n                    if mask.sum() > 0:\n                        rmsd = calculate_rmsd(pred[mask], truth[mask])\n                        sample_scores.append(rmsd)\n                    else:\n                        sample_scores.append(999.0)\n                \n                best_rmsd = min(sample_scores)\n                scores.append(best_rmsd)\n\n    avg_score = np.mean(scores)\n    print(f\"\\n>>> Validation Complete\")\n    print(f\">>> Mean Best-of-5 RMSD: {avg_score:.4f} Å (Lower is better)\")\n    return avg_score\n\n\nif __name__ == \"__main__\":\n\n    mock_seq = torch.randint(0, 4, (2, 10)) \n    \n    mock_truth = torch.randn(2, 10, 3) \n    \n    class MockModel(torch.nn.Module):\n        def forward(self, x):\n            return torch.randn(x.shape[0], 5, x.shape[1], 3)\n            \n    model = MockModel()\n    from torch.utils.data import TensorDataset, DataLoader\n    val_ds = TensorDataset(mock_seq, mock_truth)\n    val_loader = DataLoader(val_ds, batch_size=2)    \n    validate_model(model, val_loader, device='cpu')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-07T20:15:34.285735Z","iopub.execute_input":"2026-02-07T20:15:34.286489Z","iopub.status.idle":"2026-02-07T20:15:34.314278Z","shell.execute_reply.started":"2026-02-07T20:15:34.286445Z","shell.execute_reply":"2026-02-07T20:15:34.313683Z"}},"outputs":[],"execution_count":null}]}