{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# CELL 1: Setup - \nimport pandas as pd\nimport numpy as np\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')\n\nprint(\"=\"*50)\nprint(\"YOUR RNA TEMPLATE-BASED SOLUTION\")\nprint(\"=\"*50)\nprint(f\"Pandas version: {pd.__version__}\")\nprint(f\"NumPy version: {np.__version__}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:06.510855Z","iopub.execute_input":"2026-02-19T17:45:06.511205Z","iopub.status.idle":"2026-02-19T17:45:06.516234Z","shell.execute_reply.started":"2026-02-19T17:45:06.511169Z","shell.execute_reply":"2026-02-19T17:45:06.515431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#2 Find all data Files\nprint(\"=\"*50)\nprint(\"LOCATING COMPETITION FILES\")\nprint(\"=\"*50)\n\n#search for all CSV files \nfor root,dirs,files in os.walk('/kaggle/imput'):\n    for file in files:\n        if file.endswith('.csv'):\n            print(f\"☑️ Found:{os.path.join(root,file)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:06.517067Z","iopub.execute_input":"2026-02-19T17:45:06.517231Z","iopub.status.idle":"2026-02-19T17:45:06.532051Z","shell.execute_reply.started":"2026-02-19T17:45:06.517216Z","shell.execute_reply":"2026-02-19T17:45:06.531298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#3: Load training data - YOUR TEMPLATE LIBRARY\n\nprint(\"=\"*50)\nprint(\"LOADING TRAINING DATA\")\nprint(\"=\"*50)\n\n# Find train_sequences.csv\ntrain_seq_path = None\nfor root, dirs, files in os.walk('/kaggle/input'):\n    for file in files:\n        if file == 'train_sequences.csv':\n            train_seq_path = os.path.join(root, file)\n            break\n\nif train_seq_path:\n    train_seqs = pd.read_csv(train_seq_path)\n    print(f\"✅ Loaded training sequences: {len(train_seqs)} targets\")\n    print(train_seqs.head(3))\nelse:\n    print(\"❌ Could not find train_sequences.csv\")\n    raise FileNotFoundError(\"Training data missing\")\n\n# Find train_labels.csv (contains actual 3D coordinates)\ntrain_labels_path = None\nfor root, dirs, files in os.walk('/kaggle/input'):\n    for file in files:\n        if file == 'train_labels.csv':\n            train_labels_path = os.path.join(root, file)\n            break\n\nif train_labels_path:\n    train_labels = pd.read_csv(train_labels_path)\n    print(f\"\\n✅ Loaded training labels: {len(train_labels)} rows\")\n    print(train_labels.head(3))\nelse:\n    print(\"❌ Could not find train_labels.csv\")\n    raise FileNotFoundError(\"Training labels missing\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:06.534073Z","iopub.execute_input":"2026-02-19T17:45:06.534364Z","iopub.status.idle":"2026-02-19T17:45:17.714537Z","shell.execute_reply.started":"2026-02-19T17:45:06.534345Z","shell.execute_reply":"2026-02-19T17:45:17.713808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#4: Build template library from training data\nprint(\"=\"*50)\nprint(\"BUILDING TEMPLATE LIBRARY\")\nprint(\"=\"*50)\n\n# Create dictionary: target_id -> 3D coordinates\ntemplate_coords = {}\n\n# Group labels by target_id (extract from ID column)\ntrain_labels['target'] = train_labels['ID'].str.rsplit('_', n=1).str[0]\n\nfor target_id, group in train_labels.groupby('target'):\n    # Sort by residue number and get coordinates\n    coords = group.sort_values('resid')[['x_1', 'y_1', 'z_1']].values\n    template_coords[target_id] = coords\n    print(f\"  📌 Added template: {target_id} ({len(coords)} residues)\")\n\nprint(f\"\\n✅ Built template library with {len(template_coords)} structures\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:17.716228Z","iopub.execute_input":"2026-02-19T17:45:17.716472Z","iopub.status.idle":"2026-02-19T17:45:32.125214Z","shell.execute_reply.started":"2026-02-19T17:45:17.716452Z","shell.execute_reply":"2026-02-19T17:45:32.124412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#5: Simple sequence similarity (YOUR OWN IMPLEMENTATION)\nprint(\"=\"*50)\nprint(\"DEFINING SIMILARITY FUNCTION\")\nprint(\"=\"*50)\n\ndef calculate_similarity(seq1, seq2):\n    \"\"\"\n    Simple similarity score based on:\n    - Length difference penalty\n    - Percent identity\n    \"\"\"\n    len1, len2 = len(seq1), len(seq2)\n    \n    # Length difference penalty (max 30% difference allowed)\n    len_diff = abs(len1 - len2) / max(len1, len2)\n    if len_diff > 0.3:\n        return 0\n    \n    # Count matches\n    matches = 0\n    min_len = min(len1, len2)\n    for i in range(min_len):\n        if seq1[i] == seq2[i]:\n            matches += 1\n    \n    # Percent identity\n    pct_identity = matches / min_len\n    \n    # Final score: identity - length penalty\n    score = pct_identity - (len_diff * 0.5)\n    return max(0, score)\n\ndef find_best_templates(query_seq, train_seqs_df, template_coords, top_n=5):\n    \"\"\"\n    Find the most similar templates for a query sequence\n    \"\"\"\n    candidates = []\n    \n    for _, row in train_seqs_df.iterrows():\n        target_id = row['target_id']\n        train_seq = row['sequence']\n        \n        # Skip if no coordinates available\n        if target_id not in template_coords:\n            continue\n            \n        # Calculate similarity\n        sim = calculate_similarity(query_seq, train_seq)\n        \n        if sim > 0.3:  # Only keep reasonably similar templates\n            candidates.append({\n                'target_id': target_id,\n                'sequence': train_seq,\n                'similarity': sim,\n                'coords': template_coords[target_id]\n            })\n    \n    # Sort by similarity (highest first)\n    candidates.sort(key=lambda x: x['similarity'], reverse=True)\n    \n    return candidates[:top_n]\n\nprint(\"✅ Similarity function defined\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:32.126544Z","iopub.execute_input":"2026-02-19T17:45:32.126922Z","iopub.status.idle":"2026-02-19T17:45:32.13519Z","shell.execute_reply.started":"2026-02-19T17:45:32.126889Z","shell.execute_reply":"2026-02-19T17:45:32.134481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#6: Transfer template coordinates to query\nprint(\"=\"*50)\nprint(\"DEFINING TEMPLATE TRANSFER\")\nprint(\"=\"*50)\n\ndef transfer_coordinates(query_seq, template):\n    \"\"\"\n    Transfer coordinates from template to query sequence\n    Uses simple alignment based on common subsequences\n    \"\"\"\n    query_len = len(query_seq)\n    template_seq = template['sequence']\n    template_coords = template['coords']\n    \n    # Initialize with zeros\n    new_coords = np.zeros((query_len, 3))\n    \n    # Find matching subsequences\n    i = 0\n    while i < query_len:\n        best_match_len = 0\n        best_match_start = -1\n        \n        # Look for the longest matching substring\n        for j in range(len(template_seq)):\n            match_len = 0\n            while (i + match_len < query_len and \n                   j + match_len < len(template_seq) and\n                   query_seq[i + match_len] == template_seq[j + match_len]):\n                match_len += 1\n            \n            if match_len > best_match_len:\n                best_match_len = match_len\n                best_match_start = j\n        \n        # Transfer coordinates for the matched region\n        if best_match_len > 0:\n            for k in range(best_match_len):\n                if i + k < query_len:\n                    new_coords[i + k] = template_coords[best_match_start + k]\n            i += best_match_len\n        else:\n            # No match, use interpolated coordinates\n            if i > 0:\n                new_coords[i] = new_coords[i-1] + [5.0, 0, 0]\n            else:\n                new_coords[i] = [i * 5.0, 0, 0]\n            i += 1\n    \n    return new_coords\n\nprint(\"✅ Template transfer defined\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:32.136487Z","iopub.execute_input":"2026-02-19T17:45:32.136873Z","iopub.status.idle":"2026-02-19T17:45:32.154018Z","shell.execute_reply.started":"2026-02-19T17:45:32.136849Z","shell.execute_reply":"2026-02-19T17:45:32.153323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#7: Generate 5 diverse predictions\nprint(\"=\"*50)\nprint(\"DEFINING DIVERSITY FUNCTIONS\")\nprint(\"=\"*50)\n\ndef add_variation(coords, variation_type, seed):\n    \"\"\"\n    Add different types of variations to create 5 distinct predictions\n    \"\"\"\n    np.random.seed(seed)\n    result = coords.copy()\n    \n    if variation_type == 0:  # Original\n        return result\n        \n    elif variation_type == 1:  # Small random noise\n        noise = np.random.normal(0, 0.3, coords.shape)\n        result += noise\n        \n    elif variation_type == 2:  # Gentle bending\n        center = coords.mean(axis=0)\n        for i in range(len(coords)):\n            factor = 1 + 0.1 * np.sin(i * 0.1)\n            result[i] = center + (coords[i] - center) * factor\n            \n    elif variation_type == 3:  # Rotation\n        angle = np.radians(15)\n        rot_matrix = np.array([\n            [np.cos(angle), -np.sin(angle), 0],\n            [np.sin(angle), np.cos(angle), 0],\n            [0, 0, 1]\n        ])\n        result = result @ rot_matrix.T\n        \n    elif variation_type == 4:  # Stretch\n        stretch_factor = 1.1\n        result[:, 2] *= stretch_factor  # Stretch along z-axis\n    \n    return result\n\nprint(\"✅ Diversity functions defined\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:32.154866Z","iopub.execute_input":"2026-02-19T17:45:32.155145Z","iopub.status.idle":"2026-02-19T17:45:32.168386Z","shell.execute_reply.started":"2026-02-19T17:45:32.155127Z","shell.execute_reply":"2026-02-19T17:45:32.1678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#8: Load test sequences\nprint(\"=\"*50)\nprint(\"LOADING TEST SEQUENCES\")\nprint(\"=\"*50)\n\n# Find test_sequences.csv\ntest_seq_path = None\nfor root, dirs, files in os.walk('/kaggle/input'):\n    for file in files:\n        if file == 'test_sequences.csv':\n            test_seq_path = os.path.join(root, file)\n            break\n\nif test_seq_path:\n    test_seqs = pd.read_csv(test_seq_path)\n    print(f\"✅ Loaded {len(test_seqs)} test sequences\")\n    print(test_seqs.head(3))\nelse:\n    print(\"❌ Could not find test_sequences.csv\")\n    raise FileNotFoundError(\"Test data missing\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:32.169332Z","iopub.execute_input":"2026-02-19T17:45:32.169539Z","iopub.status.idle":"2026-02-19T17:45:36.249055Z","shell.execute_reply.started":"2026-02-19T17:45:32.169522Z","shell.execute_reply":"2026-02-19T17:45:36.248338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#9: Generate predictions for all test targets\nprint(\"=\"*50)\nprint(\"GENERATING PREDICTIONS\")\nprint(\"=\"*50)\n\nall_rows = []\ntotal_targets = len(test_seqs)\n\nfor idx, (_, test_row) in enumerate(test_seqs.iterrows()):\n    target_id = test_row['target_id']\n    sequence = test_row['sequence']\n    \n    print(f\"\\n📊 Processing {idx+1}/{total_targets}: {target_id} (length: {len(sequence)})\")\n    \n    # Find best templates\n    templates = find_best_templates(sequence, train_seqs, template_coords, top_n=5)\n    print(f\"   Found {len(templates)} templates\")\n    \n    # Generate 5 predictions\n    predictions = []\n    \n    for pred_num in range(1, 6):\n        if templates and pred_num <= len(templates):\n            # Use a different template for each prediction\n            template = templates[pred_num - 1]\n            print(f\"     Prediction {pred_num}: using template {template['target_id']} (similarity: {template['similarity']:.3f})\")\n            base_coords = transfer_coordinates(sequence, template)\n        else:\n            # Fallback to helix if not enough templates\n            print(f\"     Prediction {pred_num}: using fallback helix\")\n            base_coords = np.zeros((len(sequence), 3))\n            for i in range(len(sequence)):\n                angle = np.radians(i * 32.7)\n                base_coords[i] = [8.0 * np.cos(angle), 8.0 * np.sin(angle), i * 2.8]\n        \n        # Add variation for diversity\n        final_coords = add_variation(base_coords, pred_num - 1, hash(f\"{target_id}_{pred_num}\") % 10000)\n        predictions.append(final_coords)\n    \n    # Create rows for submission\n    for res_idx, nt in enumerate(sequence):\n        row_data = {\n            'ID': f\"{target_id}_1\",  # Will be overwritten\n            'resname': nt,\n            'resid': res_idx + 1\n        }\n        \n        # Add coordinates for all 5 predictions\n        for pred_num in range(5):\n            coords = predictions[pred_num][res_idx]\n            row_data[f'x_{pred_num+1}'] = coords[0]\n            row_data[f'y_{pred_num+1}'] = coords[1]\n            row_data[f'z_{pred_num+1}'] = coords[2]\n        \n        all_rows.append(row_data)\n    \n    print(f\"   ✅ Generated 5 predictions\")\n\nprint(f\"\\n✅ Total rows generated: {len(all_rows)}\")\n# After creating all_rows, add this validation\nprint(f\"\\n✅ Generated {len(all_rows)} rows\")\n\n# Quick check for any None values\nfor i, row in enumerate(all_rows[:10]):  # Check first 10\n    for key, value in row.items():\n        if value is None or pd.isna(value):\n            print(f\"⚠️ Warning: Row {i}, column {key} is None/NaN\")\n            # Fill with default value\n            row[key] = 0.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:36.250291Z","iopub.execute_input":"2026-02-19T17:45:36.250545Z","iopub.status.idle":"2026-02-19T17:45:45.06581Z","shell.execute_reply.started":"2026-02-19T17:45:36.250523Z","shell.execute_reply":"2026-02-19T17:45:45.065001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#10: Create submission DataFrame\nprint(\"=\"*50)\nprint(\"CREATING SUBMISSION DATAFRAME\")\nprint(\"=\"*50)\n\n# Convert to DataFrame\ndf = pd.DataFrame(all_rows)\n\n# Define exact column order\ncolumns = ['ID', 'resname', 'resid']\nfor i in range(1, 6):\n    columns.extend([f'x_{i}', f'y_{i}', f'z_{i}'])\n\n# Ensure all columns exist\ndf = df[columns]\n\nprint(f\"✅ DataFrame shape: {df.shape}\")\nprint(f\"✅ Columns: {list(df.columns)}\")\nprint(\"\\nFirst 3 rows:\")\nprint(df.head(3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:45.06822Z","iopub.execute_input":"2026-02-19T17:45:45.068545Z","iopub.status.idle":"2026-02-19T17:45:45.120334Z","shell.execute_reply.started":"2026-02-19T17:45:45.068523Z","shell.execute_reply":"2026-02-19T17:45:45.119636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#10.5: DEBUG MISSING VALUES\nprint(\"=\"*50)\nprint(\"DEBUGGING MISSING VALUES\")\nprint(\"=\"*50)\n\n# Check which columns have missing values\nmissing_per_column = df.isnull().sum()\nprint(\"\\n📊 Missing values per column:\")\nprint(missing_per_column[missing_per_column > 0])\n\n# Check a sample row with missing values\nif missing_per_column.sum() > 0:\n    # Find first row with any missing value\n    mask = df.isnull().any(axis=1)\n    first_missing_idx = df[mask].index[0]\n    print(f\"\\n🔍 First row with missing values (index {first_missing_idx}):\")\n    print(df.loc[first_missing_idx])\n    \n    # Show surrounding rows\n    start = max(0, first_missing_idx - 2)\n    end = min(len(df), first_missing_idx + 3)\n    print(f\"\\n📋 Rows {start} to {end-1}:\")\n    print(df.iloc[start:end])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:45.121319Z","iopub.execute_input":"2026-02-19T17:45:45.121844Z","iopub.status.idle":"2026-02-19T17:45:45.138154Z","shell.execute_reply.started":"2026-02-19T17:45:45.121816Z","shell.execute_reply":"2026-02-19T17:45:45.137544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#10.6: FILL MISSING VALUES\nprint(\"=\"*50)\nprint(\"FILLING MISSING VALUES\")\nprint(\"=\"*50)\n\n# Count missing before\nbefore = df.isnull().sum().sum()\nprint(f\"Missing values before: {before}\")\n\n# Fill all missing values with 0.0\ndf = df.fillna(0.0)\n\n# Count missing after\nafter = df.isnull().sum().sum()\nprint(f\"Missing values after: {after}\")\n\nif before > 0 and after == 0:\n    print(\"✅ All missing values filled with 0.0\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:45.138991Z","iopub.execute_input":"2026-02-19T17:45:45.139216Z","iopub.status.idle":"2026-02-19T17:45:45.150586Z","shell.execute_reply.started":"2026-02-19T17:45:45.139198Z","shell.execute_reply":"2026-02-19T17:45:45.14992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#11: Validate and save submission\nprint(\"=\"*50)\nprint(\"VALIDATING AND SAVING\")\nprint(\"=\"*50)\n\n# Check for missing values\nmissing = df.isnull().sum().sum()\nif missing == 0:\n    print(\"✅ No missing values\")\nelse:\n    print(f\"❌ Found {missing} missing values\")\n\n# Check column count\nif len(df.columns) == 18:\n    print(\"✅ Correct number of columns: 18\")\nelse:\n    print(f\"❌ Wrong columns: {len(df.columns)} (expected 18)\")\n\n# Save\noutput_path = '/kaggle/working/submission.csv'\ndf.to_csv(output_path, index=False)\n\nif os.path.exists(output_path):\n    file_size = os.path.getsize(output_path) / 1024\n    print(f\"\\n✅ File saved: {output_path}\")\n    print(f\"✅ File size: {file_size:.2f} KB\")\nelse:\n    print(\"❌ Failed to save\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:45.151451Z","iopub.execute_input":"2026-02-19T17:45:45.151734Z","iopub.status.idle":"2026-02-19T17:45:45.346815Z","shell.execute_reply.started":"2026-02-19T17:45:45.151707Z","shell.execute_reply":"2026-02-19T17:45:45.346201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#12: FINAL CHECK\nprint(\"=\"*50)\nprint(\"FINAL READINESS CHECK\")\nprint(\"=\"*50)\n\nif os.path.exists('/kaggle/working/submission.csv'):\n    print(\"✅ submission.csv is ready!\")\n    print(f\"✅ File size: {os.path.getsize('/kaggle/working/submission.csv') / 1024:.2f} KB\")\n    print(f\"✅ Rows: {len(df)}\")\n    print(f\"✅ Columns: 18\")\n    \n    print(\"\\n\" + \"=\"*50)\n    print(\"✅✅✅ READY TO SUBMIT! ✅✅✅\")\n    print(\"=\"*50)\n    print(\"\\n📋 SUBMIT NOW:\")\n    print(\"1. Click 'Save Version' (top right)\")\n    print(\"2. Click 'Submit to Competition'\")\n    print(\"3. Select 'submission.csv'\")\n    print(\"4. Click Submit\")\n    print(\"\\n⚠️ REMINDER: Internet must be OFF in Settings\")\nelse:\n    print(\"❌ submission.csv not found!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:45.347655Z","iopub.execute_input":"2026-02-19T17:45:45.347909Z","iopub.status.idle":"2026-02-19T17:45:45.353714Z","shell.execute_reply.started":"2026-02-19T17:45:45.347889Z","shell.execute_reply":"2026-02-19T17:45:45.353008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================================\n# CELL 13: ULTIMATE DIAGNOSTIC\n# ============================================\nprint(\"=\"*50)\nprint(\"ULTIMATE DIAGNOSTIC\")\nprint(\"=\"*50)\n\ndf = pd.read_csv('/kaggle/working/submission.csv')\n\nprint(f\"File exists: {os.path.exists('/kaggle/working/submission.csv')}\")\nprint(f\"File size: {os.path.getsize('/kaggle/working/submission.csv')} bytes\")\nprint(f\"Number of rows: {len(df)}\")\nprint(f\"Number of columns: {len(df.columns)}\")\nprint(f\"Column names: {list(df.columns)}\")\n\n# Check data types of coordinate columns\nprint(\"\\n📊 Data types of first few columns:\")\nfor col in df.columns[:6]:\n    print(f\"  {col}: {df[col].dtype}\")\n\n# Check for any non-numeric values in coordinate columns\ncoord_cols = [c for c in df.columns if c.startswith(('x_', 'y_', 'z_'))]\nfor col in coord_cols[:3]:\n    non_numeric = df[col].apply(lambda x: not isinstance(x, (int, float))).sum()\n    print(f\"Non-numeric in {col}: {non_numeric}\")\n\n# Show first row as dictionary\nprint(\"\\n📋 First row (raw):\")\nprint(df.iloc[0].to_dict())\n\n# Save a tiny sample for manual inspection\ndf.head(100).to_csv('/kaggle/working/sample_debug.csv', index=False)\nprint(\"\\n✅ Saved sample_debug.csv for manual inspection\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-19T17:45:45.354646Z","iopub.execute_input":"2026-02-19T17:45:45.355138Z","iopub.status.idle":"2026-02-19T17:45:45.404743Z","shell.execute_reply.started":"2026-02-19T17:45:45.355109Z","shell.execute_reply":"2026-02-19T17:45:45.404019Z"}},"outputs":[],"execution_count":null}]}