{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":118765,"databundleVersionId":15231210,"sourceType":"competition"}],"dockerImageVersionId":31239,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-08T03:48:54.593903Z","iopub.execute_input":"2026-01-08T03:48:54.594406Z","iopub.status.idle":"2026-01-08T03:49:21.792487Z","shell.execute_reply.started":"2026-01-08T03:48:54.594365Z","shell.execute_reply":"2026-01-08T03:49:21.79121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\n# --- 1. CONFIGURATION ---\nDATA_DIR = \"/kaggle/input/stanford-rna-3d-folding-2\" # Update path based on environment\nSUBMISSION_FILE = \"submission.csv\"\nTEST_FILE = os.path.join(DATA_DIR, \"test_sequences.csv\")\nMSA_DIR = os.path.join(DATA_DIR, \"MSA\")\n\n# --- 2. MODEL DEFINITION ---\nclass RNAStructurePredictor:\n    \"\"\"\n    Placeholder for your RNA Folding Model.\n    In a real scenario, this would load weights from a pre-trained model \n    and use the MSA and Sequence as input features.\n    \"\"\"\n    def predict(self, sequence, msa_path=None):\n        # sequence: string of ACGU\n        # msa_path: path to the .fasta MSA file for this target\n        L = len(sequence)\n        \n        # We need to return 5 sets of coordinates (5 predictions)\n        # Shape: (5, L, 3) -> 5 models, L residues, (x, y, z)\n        # Here we initialize with zeros or random numbers as a baseline\n        preds = np.random.normal(loc=0, scale=10, size=(5, L, 3))\n        \n        # NOTE: In a real model, you'd process the MSA here:\n        # if msa_path and os.path.exists(msa_path):\n        #     msa_data = load_msa(msa_path)\n        #     preds = my_model.inference(sequence, msa_data)\n        \n        return preds\n\n# --- 3. PIPELINE EXECUTION ---\ndef run_inference():\n    # Check if test file exists (Kaggle hidden test set)\n    if not os.path.exists(TEST_FILE):\n        # Fallback for local testing or validation\n        TEST_FILE_ALT = os.path.join(DATA_DIR, \"validation_sequences.csv\")\n        test_df = pd.read_csv(TEST_FILE_ALT)\n    else:\n        test_df = pd.read_csv(TEST_FILE)\n\n    predictor = RNAStructurePredictor()\n    all_submission_rows = []\n\n    print(f\"Starting inference on {len(test_df)} sequences...\")\n\n    for _, row in tqdm(test_df.iterrows(), total=len(test_df)):\n        target_id = row['target_id']\n        sequence = row['sequence']\n        msa_path = os.path.join(MSA_DIR, f\"{target_id}.MSA.fasta\")\n        \n        # Generate 5 structures for the entire sequence\n        # shape: (5, L, 3)\n        coords_5_models = predictor.predict(sequence, msa_path)\n        \n        # Unpack predictions into the CSV format\n        # Format: ID, resname, resid, x_1, y_1, z_1, ..., x_5, y_5, z_5\n        for i, resname in enumerate(sequence):\n            resid = i + 1\n            row_id = f\"{target_id}_{resid}\"\n            \n            # Extract (x,y,z) for this specific residue across all 5 models\n            res_coords = []\n            for m_idx in range(5):\n                x, y, z = coords_5_models[m_idx, i, :]\n                res_coords.extend([x, y, z])\n            \n            all_submission_rows.append([row_id, resname, resid] + res_coords)\n\n    # --- 4. FORMATTING & CLIPPING ---\n    columns = ['ID', 'resname', 'resid']\n    for i in range(1, 6):\n        columns += [f'x_{i}', f'y_{i}', f'z_{i}']\n\n    sub_df = pd.DataFrame(all_submission_rows, columns=columns)\n\n    # CRITICAL: Clip coordinates to prevent PDB format errors (-999.999 to 9999.999)\n    coord_cols = [c for c in sub_df.columns if c.startswith(('x_', 'y_', 'z_'))]\n    sub_df[coord_cols] = sub_df[coord_cols].clip(lower=-999.999, upper=9999.999)\n\n    # Save\n    sub_df.to_csv(SUBMISSION_FILE, index=False)\n    print(f\"Submission saved to {SUBMISSION_FILE}\")\n\nif __name__ == \"__main__\":\n    run_inference()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-08T03:50:46.937444Z","iopub.execute_input":"2026-01-08T03:50:46.938074Z","iopub.status.idle":"2026-01-08T03:50:47.48582Z","shell.execute_reply.started":"2026-01-08T03:50:46.938044Z","shell.execute_reply":"2026-01-08T03:50:47.484858Z"}},"outputs":[],"execution_count":null}]}