{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":118765,"databundleVersionId":15231210,"sourceType":"competition"},{"sourceId":14445760,"sourceType":"datasetVersion","datasetId":9227456}],"dockerImageVersionId":31234,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Conv1D, MaxPooling1D, Flatten, Dense, Embedding, Bidirectional\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom tensorflow.keras.preprocessing.text import Tokenizer\n\n# --- Randall Lujan: Decisive Creation Date: January 18, 2026 ---\n\n# 1. Load the Competition Data\n# In Kaggle, the data is usually in /kaggle/input/stanford-rna-3d-folding-2/\ntry:\n    test_df = pd.read_csv('/kaggle/input/stanford-rna-3d-folding-2/test_sequences.csv')\nexcept:\n    # Placeholder for local testing\n    test_df = pd.DataFrame({'target_id': ['R1107'], 'sequence': ['GGAAUU']})\n\n# 2. Configuration & Somatic Parameters\nMAX_LEN = 200  # Adjust based on the competition's max sequence length\nVOCAB_SIZE = 5 # A, C, G, U + Padding\n\n# 3. Hybrid CNN-LSTM Model (From your Untitled document.pdf logic)\ndef create_graei_competition_model():\n    model = Sequential([\n        Embedding(input_dim=VOCAB_SIZE, output_dim=128, input_length=MAX_LEN),\n        Conv1D(128, 5, activation='relu', padding='same'),\n        MaxPooling1D(pool_size=2),\n        Bidirectional(LSTM(128, return_sequences=True)),\n        Flatten(),\n        Dense(256, activation='relu'),\n        # Output layer: 3 coordinates (x,y,z) for every residue up to MAX_LEN\n        Dense(MAX_LEN * 3) \n    ])\n    model.compile(optimizer='adam', loss='mse')\n    return model\n\n# 4. Emotional AI Variation Logic (From Happiness, depression and euphoria.pdf)\ndef get_emotional_variance(struct_index):\n    \"\"\"\n    Spreads the 5 predictions across your emotional spectrum to maximize TM-score.\n    1: Depressed (Conservative/Stable)\n    2-3: Content (Balanced)\n    4-5: Euphoric (High-Risk/Creative)\n    \"\"\"\n    scales = {1: 0.01, 2: 0.03, 3: 0.05, 4: 0.10, 5: 0.20}\n    return scales.get(struct_index, 0.05)\n\n# 5. Generate Submission File\nmodel = create_graei_competition_model()\ntokenizer = Tokenizer(char_level=True)\ntokenizer.fit_on_texts(['A', 'C', 'G', 'U'])\n\nsubmission_data = []\n\n# Process test sequences\nfor _, row in test_df.iterrows():\n    target_id = row['target_id']\n    sequence = row['sequence']\n    \n    # Tokenize and predict\n    seq_encoded = pad_sequences(tokenizer.texts_to_sequences([sequence]), maxlen=MAX_LEN)\n    raw_coords = model.predict(seq_encoded).reshape(MAX_LEN, 3)\n    \n    # Generate the 5 required structures\n    for s_idx in range(1, 6):\n        variance = get_emotional_variance(s_idx)\n        # Apply noise based on the Emotional Scale\n        structure_coords = raw_coords + np.random.normal(0, variance, raw_coords.shape)\n        \n        for res_idx, res_name in enumerate(sequence):\n            if res_idx < MAX_LEN:\n                # Store coordinates for each of the 5 submission columns\n                submission_data.append({\n                    'ID': f\"{target_id}\",\n                    'resname': res_name,\n                    'resid': res_idx + 1,\n                    f'x_{s_idx}': structure_coords[res_idx, 0],\n                    f'y_{s_idx}': structure_coords[res_idx, 1],\n                    f'z_{s_idx}': structure_coords[res_idx, 2]\n                })\n\n# Pivot data to match the Kaggle required format: \n# ID, resname, resid, x_1, y_1, z_1, ... x_5, y_5, z_5\nfinal_df = pd.DataFrame(submission_data)\n# Aggregate the variations into a single row per residue\nfinal_submission = final_df.groupby(['ID', 'resname', 'resid']).first().reset_index()\n\n# 6. Save to CSV\nfinal_submission.to_csv('submission.csv', index=False)\nprint(\"Submission file 'submission.csv' generated successfully.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}