{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":118765,"databundleVersionId":15231210,"sourceType":"competition"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nimport pandas as pd\nimport numpy as np\nimport os\nimport json\nfrom pathlib import Path\n\nprint(\"=\" * 80)\nprint(\"STANFORD RNA 3D FOLDING 2 - DATA EXPLORATION\")\nprint(\"=\" * 80)\n\n# Set up paths\ndata_path = Path('/kaggle/input/stanford-rna-3d-folding-2')\n\n\n\n# Try to load main data files\ntry:\n    # Common file names in Kaggle competitions\n    possible_files = ['train.csv', 'train.parquet', 'train_data.csv', \n                      'train_sequences.csv', 'train_structures.csv']\n    \n    train_df = None\n    for filename in possible_files:\n        filepath = data_path / filename\n        if filepath.exists():\n            print(f\"✓ Found: {filename}\")\n            if filename.endswith('.csv'):\n                train_df = pd.read_csv(filepath, nrows=5)\n            elif filename.endswith('.parquet'):\n                train_df = pd.read_parquet(filepath)\n                train_df = train_df.head(5)\n            \n            if train_df is not None:\n                print(f\"Shape (first 5 rows): {train_df.shape}\")\n                print(f\"Column Names and Types:\")\n                print(train_df.dtypes)\n                print(f\"First 3 rows:\")\n                print(train_df.head(3))\n                print(f\"Basic Statistics:\")\n                print(train_df.describe())\n                print(f\"Missing Values:\")\n                print(train_df.isnull().sum())\n                break\n    \n    if train_df is None:\n        print(\"  No standard training file found. Listing all CSV/Parquet files:\")\n        for ext in ['*.csv', '*.parquet']:\n            for file in data_path.rglob(ext):\n                print(f\"    - {file.relative_to(data_path)}\")\n                \nexcept Exception as e:\n    print(f\"  Error loading training data: {e}\")\n\nprint(\"3. EXPLORING TEST DATA\")\nprint(\"-\" * 80)\n\ntry:\n    possible_test_files = ['test.csv', 'test.parquet', 'test_sequences.csv']\n    \n    test_df = None\n    for filename in possible_test_files:\n        filepath = data_path / filename\n        if filepath.exists():\n            print(f\"✓ Found: {filename}\")\n            if filename.endswith('.csv'):\n                test_df = pd.read_csv(filepath, nrows=5)\n            elif filename.endswith('.parquet'):\n                test_df = pd.read_parquet(filepath)\n                test_df = test_df.head(5)\n            \n            if test_df is not None:\n                print(f\"Shape (first 5 rows): {test_df.shape}\")\n                print(f\"Column Names:\")\n                print(test_df.columns.tolist())\n                print(f\"First 3 rows:\")\n                print(test_df.head(3))\n                break\n                \nexcept Exception as e:\n    print(f\"  Error loading test data: {e}\")\n\nprint(\"4. CHECKING FOR SAMPLE SUBMISSION\")\nprint(\"-\" * 80)\n\ntry:\n    sample_sub_path = data_path / 'sample_submission.csv'\n    if sample_sub_path.exists():\n        sample_sub = pd.read_csv(sample_sub_path, nrows=10)\n        print(f\"✓ Found sample_submission.csv\")\n        print(f\"Shape (first 10 rows): {sample_sub.shape}\")\n        print(f\"Columns: {sample_sub.columns.tolist()}\")\n        print(f\"First 5 rows:\")\n        print(sample_sub.head())\n        print(f\"This shows the expected submission format!\")\n    else:\n        print(\"  sample_submission.csv not found\")\nexcept Exception as e:\n    print(f\"  Error loading sample submission: {e}\")\n\nprint(\"5. CHECKING FOR ADDITIONAL DATA FILES\")\nprint(\"-\" * 80)\n\ntry:\n    # Look for structure files, metadata, etc.\n    additional_files = ['metadata.json', 'structures.csv', 'sequences.fasta', \n                       'bpp_matrices.npy', 'train_labels.csv']\n    \n    for filename in additional_files:\n        filepath = data_path / filename\n        if filepath.exists():\n            print(f\"✓ Found: {filename}\")\n            if filename.endswith('.json'):\n                with open(filepath, 'r') as f:\n                    data = json.load(f)\n                    print(f\"  Keys: {list(data.keys())[:10]}\")\n            elif filename.endswith('.csv'):\n                df = pd.read_csv(filepath, nrows=3)\n                print(f\"  Columns: {df.columns.tolist()}\")\n                \nexcept Exception as e:\n    print(f\"  Error checking additional files: {e}\")\n\nprint(\"6. SUMMARY OF KEY FINDINGS\")\nprint(\"-\" * 80)\nprint(\"\"\"Please share the output above, particularly:\n1. All available files and their sizes\n2. Training data columns and sample rows\n3. Test data structure\n4. Sample submission format\n5. Any additional data files (BPP matrices, metadata, etc.)\n\nThis will help me design the optimal model architecture!\n\"\"\")\n\nprint(\"\" + \"=\" * 80)\nprint(\"EXPLORATION COMPLETE\")\nprint(\"=\" * 80)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T14:20:27.285349Z","iopub.execute_input":"2026-01-10T14:20:27.285572Z","iopub.status.idle":"2026-01-10T14:20:29.473912Z","shell.execute_reply.started":"2026-01-10T14:20:27.285551Z","shell.execute_reply":"2026-01-10T14:20:29.472885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nfrom pathlib import Path\n\nprint(\"=\" * 80)\nprint(\"CHECKING AVAILABLE FILES\")\nprint(\"=\" * 80)\n\n# Check what's in the input directory\nDATA_PATH = Path('/kaggle/input/stanford-rna-3d-folding-2')\n\nprint(f\"📁 Contents of {DATA_PATH}:\")\nif DATA_PATH.exists():\n    for item in sorted(DATA_PATH.iterdir()):\n        if item.is_file():\n            size_mb = item.stat().st_size / (1024 * 1024)\n            print(f\"  📄 {item.name} ({size_mb:.2f} MB)\")\n        elif item.is_dir():\n            print(f\"  📁 {item.name}/\")\n            # List contents of subdirectories\n            for subitem in sorted(item.iterdir())[:10]:  # Show first 10 items\n                if subitem.is_file():\n                    size_mb = subitem.stat().st_size / (1024 * 1024)\n                    print(f\"      📄 {subitem.name} ({size_mb:.2f} MB)\")\nelse:\n    print(f\"  ❌ Directory does not exist!\")\n    print(f\"Available input directories:\")\n    input_base = Path('/kaggle/input')\n    if input_base.exists():\n        for item in sorted(input_base.iterdir()):\n            print(f\"    📁 {item.name}/\")\n\nprint(\"\" + \"=\" * 80)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T03:48:38.064476Z","iopub.execute_input":"2026-01-10T03:48:38.065185Z","iopub.status.idle":"2026-01-10T03:48:38.336346Z","shell.execute_reply.started":"2026-01-10T03:48:38.065156Z","shell.execute_reply":"2026-01-10T03:48:38.335784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Data preparation_1","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\"\"\"\nStanford RNA 3D Folding - Step 1: Data Preprocessing (FIXED ID MATCHING)\nLoads raw competition data, processes sequences and structures, creates features\n\"\"\"\n\nimport pandas as pd\nimport numpy as np\nimport pickle\nfrom pathlib import Path\nfrom tqdm import tqdm\n\nprint(\"=\" * 80)\nprint(\"STEP 1: DATA PREPROCESSING\")\nprint(\"=\" * 80)\n\n# ============================================================================\n# 1. SETUP PATHS\n# ============================================================================\nprint(\"1. Setting up paths...\")\n\nDATA_PATH = Path('/kaggle/input/stanford-rna-3d-folding-2')\nOUTPUT_PATH = Path('/kaggle/working')\nOUTPUT_PATH.mkdir(exist_ok=True)\n\nprint(f\"✓ Data path: {DATA_PATH}\")\nprint(f\"✓ Output path: {OUTPUT_PATH}\")\n\n# ============================================================================\n# 2. LOAD RAW DATA\n# ============================================================================\nprint(\"2. Loading raw competition data...\")\n\n# Load training sequences\ntrain_sequences_df = pd.read_csv(DATA_PATH / 'train_sequences.csv')\nprint(f\"✓ Loaded train_sequences.csv: {len(train_sequences_df)} structures\")\n\n# Load training labels (3D coordinates)\ntrain_labels_df = pd.read_csv(DATA_PATH / 'train_labels.csv', low_memory=False)\nprint(f\"✓ Loaded train_labels.csv: {len(train_labels_df)} coordinate entries\")\n\n# Display sample data\nprint(f\" Sample data:\")\nprint(f\"  Sequences sample:\")\nprint(train_sequences_df[['target_id', 'sequence']].head(3))\nprint(f\"Labels sample:\")\nprint(train_labels_df[['ID', 'resname', 'resid', 'x_1', 'y_1', 'z_1']].head(5))\n\n# ============================================================================\n# 3. UNDERSTAND ID STRUCTURE\n# ============================================================================\nprint(\"3. Analyzing ID structure...\")\n\n# Extract base structure ID from labels\n# Labels have format like \"157D_1\", \"157D_2\" where the suffix is the residue number\n# We need to extract just \"157D\" to match with target_id\n\n# Check unique ID patterns\nsample_label_ids = train_labels_df['ID'].head(20).tolist()\nprint(f\"  Sample label IDs: {sample_label_ids[:5]}\")\n\n# Extract base structure IDs from labels (remove suffix after underscore)\ntrain_labels_df['structure_id'] = train_labels_df['ID'].str.rsplit('_', n=1).str[0]\n\n# Check what we got\nunique_structure_ids = train_labels_df['structure_id'].unique()\nprint(f\"  Unique structure IDs in labels: {len(unique_structure_ids)}\")\nprint(f\"  Sample structure IDs: {list(unique_structure_ids[:5])}\")\n\n# Check target_ids from sequences\nunique_target_ids = train_sequences_df['target_id'].unique()\nprint(f\"  Unique target IDs in sequences: {len(unique_target_ids)}\")\nprint(f\"  Sample target IDs: {list(unique_target_ids[:5])}\")\n\n# Find overlap\noverlap = set(unique_structure_ids) & set(unique_target_ids)\nprint(f\"  ✅ Overlapping IDs: {len(overlap)}\")\n\nif len(overlap) == 0:\n    print(\" WARNING: No overlap found! Checking alternative ID formats...\")\n    # Try different extraction methods\n    print(f\"  Trying to match case-insensitive...\")\n    structure_ids_lower = set(s.lower() for s in unique_structure_ids)\n    target_ids_lower = set(t.lower() for t in unique_target_ids)\n    overlap_lower = structure_ids_lower & target_ids_lower\n    print(f\"  Case-insensitive overlap: {len(overlap_lower)}\")\n\n# ============================================================================\n# 4. PROCESS SEQUENCES AND COORDINATES\n# ============================================================================\nprint(\"4. Processing sequences and coordinates...\")\n\n# Group labels by structure ID\nlabels_grouped = train_labels_df.groupby('structure_id')\n\n# Create dictionaries for sequences and coordinates\nsequences = {}\ncoords_dict = {}\n\nprint(\"Processing training data...\")\nfor _, row in tqdm(train_sequences_df.iterrows(), total=len(train_sequences_df)):\n    target_id = row['target_id']\n    sequence = row['sequence']\n    sequences[target_id] = sequence\n    \n    # Get corresponding coordinates\n    if target_id in labels_grouped.groups:\n        label_rows = labels_grouped.get_group(target_id)\n        # Sort by residue ID to ensure correct order\n        label_rows = label_rows.sort_values('resid')\n        # Extract x, y, z coordinates\n        coords = label_rows[['x_1', 'y_1', 'z_1']].values\n        coords_dict[target_id] = coords\n\nprint(f\"✓ Processed {len(sequences)} sequences\")\nprint(f\"✓ Processed {len(coords_dict)} coordinate sets\")\n\n# ============================================================================\n# 5. CREATE FEATURE ENCODINGS\n# ============================================================================\nprint(\"5. Creating feature encodings...\")\n\ndef one_hot_encode_sequence(sequence):\n    \"\"\"\n    One-hot encode RNA sequence\n    A, C, G, U → 4-dimensional vectors\n    \"\"\"\n    mapping = {'A': [1, 0, 0, 0],\n               'C': [0, 1, 0, 0],\n               'G': [0, 0, 1, 0],\n               'U': [0, 0, 0, 1],\n               'T': [0, 0, 0, 1]}  # Treat T as U\n    \n    encoded = []\n    for nucleotide in sequence.upper():\n        if nucleotide in mapping:\n            encoded.append(mapping[nucleotide])\n        else:\n            # Unknown nucleotide → all zeros\n            encoded.append([0, 0, 0, 0])\n    \n    return np.array(encoded)\n\ndef positional_encoding(seq_len, d_model=4):\n    \"\"\"\n    Create sinusoidal positional encoding\n    Helps model understand position in sequence\n    \"\"\"\n    position = np.arange(seq_len)[:, np.newaxis]\n    div_term = np.exp(np.arange(0, d_model, 2) * -(np.log(10000.0) / d_model))\n    \n    pos_encoding = np.zeros((seq_len, d_model))\n    pos_encoding[:, 0::2] = np.sin(position * div_term)\n    pos_encoding[:, 1::2] = np.cos(position * div_term)\n    \n    return pos_encoding\n\n# Process all sequences\ntrain_features = []\n\nprint(\"Creating feature encodings...\")\nfor target_id in tqdm(sequences.keys()):\n    sequence = sequences[target_id]\n    seq_len = len(sequence)\n    \n    # One-hot encoding (4 dimensions)\n    one_hot = one_hot_encode_sequence(sequence)\n    \n    # Positional encoding (4 dimensions)\n    pos_enc = positional_encoding(seq_len, d_model=4)\n    \n    # Combine features (8 dimensions total)\n    features = np.concatenate([one_hot, pos_enc], axis=1)\n    \n    # Store\n    train_features.append({\n        'target_id': target_id,\n        'seq_len': seq_len,\n        'features': features  # Shape: (seq_len, 8)\n    })\n\nprint(f\"✓ Created features for {len(train_features)} structures\")\n\n# ============================================================================\n# 6. CREATE LABELS DICTIONARY\n# ============================================================================\nprint(\"6. Creating labels dictionary...\")\n\nlabels_dict = {}\nfor target_id, coords in coords_dict.items():\n    labels_dict[target_id] = {\n        'coords': coords,  # Shape: (seq_len, 3) - x, y, z coordinates\n        'seq_len': len(coords)\n    }\n\nprint(f\"✓ Created labels for {len(labels_dict)} structures\")\n\n# ============================================================================\n# 7. DATA VALIDATION & CLEANING\n# ============================================================================\nprint(\"7. Validating and cleaning data...\")\n\n# Check for mismatches\nvalid_ids = set(sequences.keys()) & set(coords_dict.keys())\nprint(f\"✓ Valid structures (with both sequence and coords): {len(valid_ids)}\")\n\nif len(valid_ids) == 0:\n    print(\" ERROR: No valid structures found!\")\n    print(\"   This means no IDs match between sequences and labels.\")\n    print(\"   Debugging information:\")\n    print(f\"   - Sequences have {len(sequences)} entries\")\n    print(f\"   - Coordinates have {len(coords_dict)} entries\")\n    print(f\"   - Sample sequence IDs: {list(sequences.keys())[:5]}\")\n    print(f\"   - Sample coordinate IDs: {list(coords_dict.keys())[:5]}\")\n    raise ValueError(\"No matching IDs found between sequences and labels!\")\n\n# Check for length mismatches\nlength_mismatches = 0\nmismatch_ids = []\nfor target_id in valid_ids:\n    seq_len = len(sequences[target_id])\n    coord_len = len(coords_dict[target_id])\n    if seq_len != coord_len:\n        length_mismatches += 1\n        mismatch_ids.append((target_id, seq_len, coord_len))\n\nif length_mismatches > 0:\n    print(f\" Found {length_mismatches} structures with sequence/coordinate length mismatches\")\n    print(f\"   First 5 mismatches (ID, seq_len, coord_len):\")\n    for item in mismatch_ids[:5]:\n        print(f\"     {item}\")\nelse:\n    print(f\"✓ All structures have matching sequence/coordinate lengths\")\n\n# Check for NaN/Inf values\nbad_structures = []\ngood_structures = []\nfor target_id in valid_ids:\n    coords = coords_dict[target_id]\n    if np.any(np.isnan(coords)) or np.any(np.isinf(coords)):\n        bad_structures.append(target_id)\n    else:\n        good_structures.append(target_id)\n\nprint(f\"✅ Good structures: {len(good_structures)}\")\nprint(f\"❌ Bad structures (with NaN/Inf): {len(bad_structures)}\")\n\nif len(bad_structures) > 0:\n    print(f\"   First 5 bad structures: {bad_structures[:5]}\")\n\n# Use only good structures\nvalid_ids = set(good_structures)\nlabels_dict = {k: v for k, v in labels_dict.items() if k in valid_ids}\n\n# ============================================================================\n# 8. CREATE TRAIN/VAL SPLIT\n# ============================================================================\nprint(\"8. Creating train/validation split...\")\n\n# Use 90/10 split (more data for training)\nnp.random.seed(42)\nall_ids = list(valid_ids)\nnp.random.shuffle(all_ids)\n\nsplit_idx = int(0.9 * len(all_ids))\ntrain_ids = set(all_ids[:split_idx])\nval_ids = set(all_ids[split_idx:])\n\nprint(f\"✓ Train set: {len(train_ids)} structures\")\nprint(f\"✓ Validation set: {len(val_ids)} structures\")\n\n# ============================================================================\n# 9. SAVE PROCESSED DATA\n# ============================================================================\nprint(\"9. Saving processed data...\")\n\n# Save features\nwith open(OUTPUT_PATH / 'train_features.pkl', 'wb') as f:\n    pickle.dump(train_features, f)\nprint(f\"✓ Saved train_features.pkl\")\n\n# Save labels\nwith open(OUTPUT_PATH / 'labels_dict.pkl', 'wb') as f:\n    pickle.dump(labels_dict, f)\nprint(f\"✓ Saved labels_dict.pkl\")\n\n# Save train/val split\nwith open(OUTPUT_PATH / 'train_val_split.pkl', 'wb') as f:\n    pickle.dump({'train_ids': train_ids, 'val_ids': val_ids}, f)\nprint(f\"✓ Saved train_val_split.pkl\")\n\n# ============================================================================\n# 10. SUMMARY STATISTICS\n# ============================================================================\nprint(\"\" + \"=\" * 80)\nprint(\"PREPROCESSING COMPLETE!\")\nprint(\"=\" * 80)\nprint(\" Dataset Statistics:\")\nprint(f\"  Total structures: {len(valid_ids)}\")\nprint(f\"  Train structures: {len(train_ids)}\")\nprint(f\"  Validation structures: {len(val_ids)}\")\n\n# Sequence length statistics\nseq_lengths = [len(sequences[tid]) for tid in valid_ids]\nprint(f\" Sequence Length Statistics:\")\nprint(f\"  Min length: {min(seq_lengths)} nucleotides\")\nprint(f\"  Max length: {max(seq_lengths)} nucleotides\")\nprint(f\"  Mean length: {np.mean(seq_lengths):.1f} nucleotides\")\nprint(f\"  Median length: {np.median(seq_lengths):.1f} nucleotides\")\n\n# Count sequences by length ranges\nlength_ranges = [\n    (0, 100, \"0-100\"),\n    (100, 500, \"100-500\"),\n    (500, 1000, \"500-1000\"),\n    (1000, 5000, \"1000-5000\"),\n    (5000, float('inf'), \">5000\")\n]\n\nprint(f\"Sequence Length Distribution:\")\nfor min_len, max_len, label in length_ranges:\n    count = sum(1 for l in seq_lengths if min_len <= l < max_len)\n    percentage = (count / len(seq_lengths)) * 100\n    print(f\"  {label:>10} nt: {count:>5} structures ({percentage:>5.1f}%)\")\n\nprint(\" Ready for Step 2: Model Training!\")\nprint(\"=\" * 80)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T04:03:55.389929Z","iopub.execute_input":"2026-01-10T04:03:55.390257Z","iopub.status.idle":"2026-01-10T04:04:32.974544Z","shell.execute_reply.started":"2026-01-10T04:03:55.390234Z","shell.execute_reply":"2026-01-10T04:04:32.97366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nprint(f\"CUDA available: {torch.cuda.is_available()}\")\nif torch.cuda.is_available():\n    print(f\"GPU device: {torch.cuda.get_device_name(0)}\")\n    print(f\"GPU memory: {torch.cuda.get_device_properties(0).total_memory / 1e9:.2f} GB\")\nelse:\n    print(\"⚠️ No GPU detected - please enable accelerator\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T04:16:19.308849Z","iopub.execute_input":"2026-01-10T04:16:19.309172Z","iopub.status.idle":"2026-01-10T04:16:19.341379Z","shell.execute_reply.started":"2026-01-10T04:16:19.309151Z","shell.execute_reply":"2026-01-10T04:16:19.340815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Step 2 model training ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n\"\"\"\nStanford RNA 3D Folding - Step 2: Baseline Model Training\nIncludes data cleaning, filtering, and stable training\n\"\"\"\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nimport numpy as np\nimport pickle\nfrom pathlib import Path\nfrom tqdm import tqdm\n\nprint(\"=\" * 80)\nprint(\"STEP 2: BASELINE MODEL TRAINING (MINIMAL & STABLE)\")\nprint(\"=\" * 80)\n\n# Check GPU availability\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"✓ Using device: {device}\")\n\n# ============================================================================\n# 1. LOAD PROCESSED DATA\n# ============================================================================\nprint(\"1. Loading processed data...\")\n\nOUTPUT_PATH = Path('/kaggle/working')\n\nwith open(OUTPUT_PATH / 'train_features.pkl', 'rb') as f:\n    train_features = pickle.load(f)\n\nwith open(OUTPUT_PATH / 'labels_dict.pkl', 'rb') as f:\n    labels_dict = pickle.load(f)\n\nwith open(OUTPUT_PATH / 'train_val_split.pkl', 'rb') as f:\n    splits = pickle.load(f)\n    train_ids = splits['train_ids']\n    val_ids = splits['val_ids']\n\nprint(f\"✓ Loaded {len(train_features)} feature sets\")\nprint(f\"✓ Loaded {len(labels_dict)} label sets\")\nprint(f\"✓ Train IDs: {len(train_ids)}\")\nprint(f\"✓ Val IDs: {len(val_ids)}\")\n\n# ============================================================================\n# 2. CREATE PYTORCH DATASET WITH LENGTH FILTERING\n# ============================================================================\nprint(\"2. Creating PyTorch datasets...\")\n\nclass RNADataset(Dataset):\n    \"\"\"Dataset for RNA 3D structure prediction with length filtering\"\"\"\n    \n    def __init__(self, features_list, labels_dict, target_ids, max_length=5000):\n        # Filter by target_ids AND sequence length\n        self.features_list = [\n            f for f in features_list\n            if f['target_id'] in target_ids and f['seq_len'] <= max_length\n        ]\n        self.labels_dict = labels_dict\n        self.max_length = max_length\n        \n        # Report filtering stats\n        original_count = len([f for f in features_list if f['target_id'] in target_ids])\n        filtered_count = len(self.features_list)\n        print(f\"  Filtered sequences: {original_count} → {filtered_count} (removed {original_count - filtered_count} sequences > {max_length} nt)\")\n    \n    def __len__(self):\n        return len(self.features_list)\n    \n    def __getitem__(self, idx):\n        feature_data = self.features_list[idx]\n        target_id = feature_data['target_id']\n        \n        # Get features (sequence encoding + positional encoding)\n        features = torch.FloatTensor(feature_data['features'])\n        \n        # Get labels (3D coordinates - NO NORMALIZATION)\n        labels = torch.FloatTensor(self.labels_dict[target_id]['coords'])\n        \n        # Ensure contiguous memory layout (fixes cuDNN error)\n        features = features.contiguous()\n        labels = labels.contiguous()\n        \n        return features, labels, target_id\n\n# Create datasets with length filtering\nMAX_SEQ_LENGTH = 5000  # Train only on sequences up to 5000 nucleotides\n\ntrain_dataset = RNADataset(train_features, labels_dict, train_ids, max_length=MAX_SEQ_LENGTH)\nval_dataset = RNADataset(train_features, labels_dict, val_ids, max_length=MAX_SEQ_LENGTH)\n\nprint(f\"✓ Train dataset: {len(train_dataset)} samples\")\nprint(f\"✓ Val dataset: {len(val_dataset)} samples\")\n\n# ============================================================================\n# 3. DEFINE MODEL ARCHITECTURE\n# ============================================================================\nprint(\"3. Defining model architecture...\")\n\nclass RNAStructurePredictor(nn.Module):\n    \"\"\"\n    Simple LSTM-based model for RNA 3D structure prediction\n    \n    Architecture:\n    - Input: Sequence features (one-hot + positional encoding)\n    - Bidirectional LSTM layers\n    - Fully connected layers\n    - Output: 3D coordinates (x, y, z) for each residue\n    \"\"\"\n    \n    def __init__(self, input_dim=8, hidden_dim=256, num_layers=2, output_dim=3):\n        super(RNAStructurePredictor, self).__init__()\n        \n        # Bidirectional LSTM\n        self.lstm = nn.LSTM(\n            input_dim,\n            hidden_dim,\n            num_layers=num_layers,\n            bidirectional=True,\n            batch_first=True,\n            dropout=0.2 if num_layers > 1 else 0\n        )\n        \n        # Fully connected layers\n        self.fc1 = nn.Linear(hidden_dim * 2, hidden_dim)  # *2 for bidirectional\n        self.relu = nn.ReLU()\n        self.dropout = nn.Dropout(0.2)\n        self.fc2 = nn.Linear(hidden_dim, output_dim)\n    \n    def forward(self, x):\n        # x shape: (batch_size, seq_len, input_dim)\n        \n        # LSTM layer\n        lstm_out, _ = self.lstm(x)\n        # lstm_out shape: (batch_size, seq_len, hidden_dim * 2)\n        \n        # Fully connected layers\n        out = self.fc1(lstm_out)\n        out = self.relu(out)\n        out = self.dropout(out)\n        out = self.fc2(out)\n        # out shape: (batch_size, seq_len, 3)\n        \n        return out\n\n# Initialize model\nmodel = RNAStructurePredictor(\n    input_dim=8,  # 4 (one-hot) + 4 (positional encoding)\n    hidden_dim=256,\n    num_layers=2,\n    output_dim=3  # x, y, z coordinates\n).to(device)\n\nprint(f\"✓ Model created with {sum(p.numel() for p in model.parameters()):,} parameters\")\n\n# ============================================================================\n# 4. TRAINING SETUP (MINIMAL & STABLE)\n# ============================================================================\nprint(\"4. Setting up training...\")\n\n# Loss function: Mean Squared Error (RMSD-like)\ncriterion = nn.MSELoss()\n\n# Optimizer: Adam with VERY LOW learning rate\noptimizer = optim.Adam(model.parameters(), lr=0.00001)  # Very conservative\n\n# Learning rate scheduler\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(\n    optimizer, mode='min', factor=0.5, patience=2\n)\n\nprint(\"✓ Loss: MSE\")\nprint(\"✓ Optimizer: Adam (lr=0.00001)\")\nprint(\"✓ Scheduler: ReduceLROnPlateau\")\nprint(\"✓ Gradient clipping: enabled (max_norm=1.0)\")\n\n# ============================================================================\n# 5. TRAINING LOOP WITH GRADIENT CLIPPING\n# ============================================================================\nprint(\"5. Training model...\")\n\ndef train_epoch(model, dataloader, criterion, optimizer, device):\n    \"\"\"Train for one epoch with gradient clipping\"\"\"\n    model.train()\n    total_loss = 0\n    nan_count = 0\n    valid_batches = 0\n    \n    for features, labels, _ in tqdm(dataloader, desc=\"Training\"):\n        features = features.to(device)\n        labels = labels.to(device)\n        \n        # Forward pass\n        optimizer.zero_grad()\n        outputs = model(features)\n        \n        # Calculate loss\n        loss = criterion(outputs, labels)\n        \n        # Check for NaN\n        if torch.isnan(loss) or torch.isinf(loss):\n            nan_count += 1\n            continue  # Skip this batch\n        \n        # Backward pass\n        loss.backward()\n        \n        # GRADIENT CLIPPING (prevents exploding gradients)\n        torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n        \n        optimizer.step()\n        \n        total_loss += loss.item()\n        valid_batches += 1\n    \n    if nan_count > 0:\n        print(f\"  ⚠️ Skipped {nan_count} batches with NaN/Inf loss\")\n    \n    return total_loss / valid_batches if valid_batches > 0 else float('inf')\n\ndef validate(model, dataloader, criterion, device):\n    \"\"\"Validate model\"\"\"\n    model.eval()\n    total_loss = 0\n    nan_count = 0\n    valid_batches = 0\n    \n    with torch.no_grad():\n        for features, labels, _ in tqdm(dataloader, desc=\"Validation\"):\n            features = features.to(device)\n            labels = labels.to(device)\n            \n            # Forward pass\n            outputs = model(features)\n            \n            # Calculate loss\n            loss = criterion(outputs, labels)\n            \n            # Check for NaN\n            if torch.isnan(loss) or torch.isinf(loss):\n                nan_count += 1\n                continue\n            \n            total_loss += loss.item()\n            valid_batches += 1\n    \n    if nan_count > 0:\n        print(f\"  ⚠️ Skipped {nan_count} batches with NaN/Inf loss\")\n    \n    return total_loss / valid_batches if valid_batches > 0 else float('inf')\n\n# Training parameters\nNUM_EPOCHS = 10\nBATCH_SIZE = 1  # Process one structure at a time (variable length)\n\n# Create dataloaders\ntrain_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=False)\n\n# Training loop\nbest_val_loss = float('inf')\ntrain_losses = []\nval_losses = []\n\nfor epoch in range(NUM_EPOCHS):\n    print(f\"Epoch {epoch + 1}/{NUM_EPOCHS}\")\n    print(\"-\" * 40)\n    \n    # Train\n    train_loss = train_epoch(model, train_loader, criterion, optimizer, device)\n    train_losses.append(train_loss)\n    \n    # Validate\n    val_loss = validate(model, val_loader, criterion, device)\n    val_losses.append(val_loss)\n    \n    # Update learning rate\n    scheduler.step(val_loss)\n    \n    print(f\"Train Loss: {train_loss:.4f}\")\n    print(f\"Val Loss: {val_loss:.4f}\")\n    \n    # Save best model\n    if val_loss < best_val_loss:\n        best_val_loss = val_loss\n        torch.save(model.state_dict(), OUTPUT_PATH / 'best_model.pth')\n        print(\"✓ Saved best model!\")\n\nprint(\"\" + \"=\" * 80)\nprint(\"TRAINING COMPLETE!\")\nprint(f\"Best validation loss: {best_val_loss:.4f}\")\nprint(\"=\" * 80)\n\nprint(\" Training Summary:\")\nprint(f\"  Initial train loss: {train_losses[0]:.4f}\")\nprint(f\"  Final train loss: {train_losses[-1]:.4f}\")\nprint(f\"  Initial val loss: {val_losses[0]:.4f}\")\nprint(f\"  Final val loss: {val_losses[-1]:.4f}\")\nprint(f\"  Best val loss: {best_val_loss:.4f}\")\nprint(f\"  Improvement: {((val_losses[0] - best_val_loss) / val_losses[0] * 100):.1f}%\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T04:17:21.610859Z","iopub.execute_input":"2026-01-10T04:17:21.61153Z","iopub.status.idle":"2026-01-10T04:31:36.491891Z","shell.execute_reply.started":"2026-01-10T04:17:21.611502Z","shell.execute_reply":"2026-01-10T04:31:36.491127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Generating the submission.csv ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\"\"\"\nStanford RNA 3D Folding - Step 3: Generate Predictions and Create Submission\nLoads trained model, processes test sequences, generates predictions, creates submission.csv\n\"\"\"\n\nimport torch\nimport torch.nn as nn\nimport pandas as pd\nimport numpy as np\nimport pickle\nfrom pathlib import Path\nfrom tqdm import tqdm\n\nprint(\"=\" * 80)\nprint(\"STEP 3: GENERATE PREDICTIONS & CREATE SUBMISSION\")\nprint(\"=\" * 80)\n\n# Check GPU availability\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"✓ Using device: {device}\")\n\n# ============================================================================\n# 1. SETUP PATHS\n# ============================================================================\nprint(\"1. Setting up paths...\")\n\nDATA_PATH = Path('/kaggle/input/stanford-rna-3d-folding-2')\nOUTPUT_PATH = Path('/kaggle/working')\n\nprint(f\"✓ Data path: {DATA_PATH}\")\nprint(f\"✓ Output path: {OUTPUT_PATH}\")\n\n# ============================================================================\n# 2. DEFINE MODEL ARCHITECTURE (SAME AS TRAINING)\n# ============================================================================\nprint(\"2. Defining model architecture...\")\n\nclass RNAStructurePredictor(nn.Module):\n    \"\"\"\n    Simple LSTM-based model for RNA 3D structure prediction\n    \"\"\"\n    \n    def __init__(self, input_dim=8, hidden_dim=256, num_layers=2, output_dim=3):\n        super(RNAStructurePredictor, self).__init__()\n        \n        # Bidirectional LSTM\n        self.lstm = nn.LSTM(\n            input_dim,\n            hidden_dim,\n            num_layers=num_layers,\n            bidirectional=True,\n            batch_first=True,\n            dropout=0.2 if num_layers > 1 else 0\n        )\n        \n        # Fully connected layers\n        self.fc1 = nn.Linear(hidden_dim * 2, hidden_dim)\n        self.relu = nn.ReLU()\n        self.dropout = nn.Dropout(0.2)\n        self.fc2 = nn.Linear(hidden_dim, output_dim)\n    \n    def forward(self, x):\n        lstm_out, _ = self.lstm(x)\n        out = self.fc1(lstm_out)\n        out = self.relu(out)\n        out = self.dropout(out)\n        out = self.fc2(out)\n        return out\n\n# Initialize model\nmodel = RNAStructurePredictor(\n    input_dim=8,\n    hidden_dim=256,\n    num_layers=2,\n    output_dim=3\n).to(device)\n\nprint(f\"✓ Model architecture defined\")\n\n# ============================================================================\n# 3. LOAD TRAINED MODEL WEIGHTS\n# ============================================================================\nprint(\"3. Loading trained model weights...\")\n\nmodel_path = OUTPUT_PATH / 'best_model.pth'\nif not model_path.exists():\n    raise FileNotFoundError(f\"❌ Model not found at {model_path}\")\n\nmodel.load_state_dict(torch.load(model_path, map_location=device))\nmodel.eval()  # Set to evaluation mode\nprint(f\"✓ Loaded model from {model_path}\")\n\n# ============================================================================\n# 4. LOAD TEST DATA\n# ============================================================================\nprint(\"4. Loading test data...\")\n\n# Load test sequences\ntest_sequences_df = pd.read_csv(DATA_PATH / 'test_sequences.csv')\nprint(f\"✓ Loaded test_sequences.csv: {len(test_sequences_df)} structures\")\n\n# Display sample\nprint(f\"Test data preview:\")\nprint(test_sequences_df.head(3))\n\n# ============================================================================\n# 5. CREATE FEATURE ENCODINGS FOR TEST DATA\n# ============================================================================\nprint(\"5. Creating feature encodings for test data...\")\n\ndef one_hot_encode_sequence(sequence):\n    \"\"\"One-hot encode RNA sequence\"\"\"\n    mapping = {'A': [1, 0, 0, 0],\n               'C': [0, 1, 0, 0],\n               'G': [0, 0, 1, 0],\n               'U': [0, 0, 0, 1],\n               'T': [0, 0, 0, 1]}\n    \n    encoded = []\n    for nucleotide in sequence.upper():\n        if nucleotide in mapping:\n            encoded.append(mapping[nucleotide])\n        else:\n            encoded.append([0, 0, 0, 0])\n    \n    return np.array(encoded)\n\ndef positional_encoding(seq_len, d_model=4):\n    \"\"\"Create sinusoidal positional encoding\"\"\"\n    position = np.arange(seq_len)[:, np.newaxis]\n    div_term = np.exp(np.arange(0, d_model, 2) * -(np.log(10000.0) / d_model))\n    \n    pos_encoding = np.zeros((seq_len, d_model))\n    pos_encoding[:, 0::2] = np.sin(position * div_term)\n    pos_encoding[:, 1::2] = np.cos(position * div_term)\n    \n    return pos_encoding\n\n# Process test sequences\ntest_features = []\n\nprint(\"Processing test sequences...\")\nfor _, row in tqdm(test_sequences_df.iterrows(), total=len(test_sequences_df)):\n    target_id = row['target_id']\n    sequence = row['sequence']\n    seq_len = len(sequence)\n    \n    # One-hot encoding (4 dimensions)\n    one_hot = one_hot_encode_sequence(sequence)\n    \n    # Positional encoding (4 dimensions)\n    pos_enc = positional_encoding(seq_len, d_model=4)\n    \n    # Combine features (8 dimensions total)\n    features = np.concatenate([one_hot, pos_enc], axis=1)\n    \n    test_features.append({\n        'target_id': target_id,\n        'seq_len': seq_len,\n        'features': features\n    })\n\nprint(f\"✓ Created features for {len(test_features)} test structures\")\n\n# ============================================================================\n# 6. GENERATE PREDICTIONS\n# ============================================================================\nprint(\"6. Generating predictions...\")\n\npredictions = {}\n\nwith torch.no_grad():\n    for feature_data in tqdm(test_features, desc=\"Predicting\"):\n        target_id = feature_data['target_id']\n        features = torch.FloatTensor(feature_data['features']).unsqueeze(0).to(device)\n        \n        # Generate prediction\n        pred_coords = model(features)\n        \n        # Convert to numpy\n        pred_coords = pred_coords.cpu().numpy()[0]  # Shape: (seq_len, 3)\n        \n        # Store predictions\n        predictions[target_id] = pred_coords\n\nprint(f\"✓ Generated predictions for {len(predictions)} structures\")\n\n# ============================================================================\n# 7. CREATE SUBMISSION FILE\n# ============================================================================\nprint(\"7. Creating submission file...\")\n\n# Load sample submission to understand format\nsample_submission = pd.read_csv(DATA_PATH / 'sample_submission.csv')\nprint(f\"✓ Loaded sample_submission.csv: {len(sample_submission)} rows\")\n\nprint(f\" Sample submission format:\")\nprint(sample_submission.head(5))\n\n# Create submission dataframe\nsubmission_rows = []\n\nprint(\"Building submission rows...\")\nfor target_id, pred_coords in tqdm(predictions.items()):\n    # Each residue gets one row\n    for residue_idx in range(len(pred_coords)):\n        # Create ID in format: target_id_residue_number\n        row_id = f\"{target_id}_{residue_idx + 1}\"\n        x, y, z = pred_coords[residue_idx]\n        \n        submission_rows.append({\n            'ID': row_id,\n            'x': x,\n            'y': y,\n            'z': z\n        })\n\n# Create DataFrame\nsubmission_df = pd.DataFrame(submission_rows)\n\n# Ensure correct column order\nsubmission_df = submission_df[['ID', 'x', 'y', 'z']]\n\nprint(f\"✓ Created submission with {len(submission_df)} rows\")\n\n# ============================================================================\n# 8. VALIDATE SUBMISSION FORMAT\n# ============================================================================\nprint(\"8. Validating submission format...\")\n\n# Check columns\nexpected_columns = ['ID', 'x', 'y', 'z']\nif list(submission_df.columns) != expected_columns:\n    print(f\"⚠️  Column mismatch!\")\n    print(f\"   Expected: {expected_columns}\")\n    print(f\"   Got: {list(submission_df.columns)}\")\nelse:\n    print(f\"✓ Columns correct: {list(submission_df.columns)}\")\n\n# Check for NaN values\nnan_count = submission_df.isna().sum().sum()\nif nan_count > 0:\n    print(f\"⚠️  Found {nan_count} NaN values in submission!\")\n    print(f\"   Replacing with 0.0...\")\n    submission_df = submission_df.fillna(0.0)\nelse:\n    print(f\"✓ No NaN values found\")\n\n# Check row count\nprint(f\"✓ Total rows: {len(submission_df)}\")\n\n# Display sample\nprint(f\"Submission preview:\")\nprint(submission_df.head(10))\n\n# ============================================================================\n# 9. SAVE SUBMISSION FILE\n# ============================================================================\nprint(\"9. Saving submission file...\")\n\nsubmission_path = OUTPUT_PATH / 'submission.csv'\nsubmission_df.to_csv(submission_path, index=False)\n\nprint(f\"✓ Saved submission to: {submission_path}\")\n\n# ============================================================================\n# 10. SUMMARY STATISTICS\n# ============================================================================\nprint(\"\" + \"=\" * 80)\nprint(\"SUBMISSION GENERATION COMPLETE!\")\nprint(\"=\" * 80)\n\nprint(\"Submission Statistics:\")\nprint(f\"  Total test structures: {len(predictions)}\")\nprint(f\"  Total prediction rows: {len(submission_df)}\")\nprint(f\"  File size: {submission_path.stat().st_size / 1024:.2f} KB\")\n\n# Coordinate statistics\nprint(f\" Coordinate Statistics:\")\nprint(f\"  X range: [{submission_df['x'].min():.2f}, {submission_df['x'].max():.2f}]\")\nprint(f\"  Y range: [{submission_df['y'].min():.2f}, {submission_df['y'].max():.2f}]\")\nprint(f\"  Z range: [{submission_df['z'].min():.2f}, {submission_df['z'].max():.2f}]\")\n\n\nprint(\"=\" * 80)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T04:40:17.473838Z","iopub.execute_input":"2026-01-10T04:40:17.474274Z","iopub.status.idle":"2026-01-10T04:40:17.989394Z","shell.execute_reply.started":"2026-01-10T04:40:17.474252Z","shell.execute_reply":"2026-01-10T04:40:17.988673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import FileLink\n\n# Create download link\nFileLink('/kaggle/working/submission.csv')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T04:43:54.863173Z","iopub.execute_input":"2026-01-10T04:43:54.863691Z","iopub.status.idle":"2026-01-10T04:43:54.869317Z","shell.execute_reply.started":"2026-01-10T04:43:54.863668Z","shell.execute_reply":"2026-01-10T04:43:54.868534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display submission content for manual copy\nimport pandas as pd\n\ndf = pd.read_csv('/kaggle/working/submission.csv')\nprint(f\"Total rows: {len(df)}\")\nprint(\"First 20 rows:\")\nprint(df.head(20).to_csv(index=False))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-10T04:59:54.457343Z","iopub.execute_input":"2026-01-10T04:59:54.458055Z","iopub.status.idle":"2026-01-10T04:59:54.472377Z","shell.execute_reply.started":"2026-01-10T04:59:54.458028Z","shell.execute_reply":"2026-01-10T04:59:54.471699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}