{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210,"isSourceIdPinned":false}],"dockerImageVersionId":31328,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nDATA_PATH = \"/kaggle/input/competitions/stanford-rna-3d-folding-2\"\n\n# Read test sequences and training labels\ntest_df   = pd.read_csv(f\"{DATA_PATH}/test_sequences.csv\")\ntrain_lab = pd.read_csv(f\"{DATA_PATH}/train_labels.csv\")\n\nprint(f\"Test sequences: {len(test_df)}\")\nprint(f\"Training labels loaded: {len(train_lab)} rows\")\n\n# Calculate average x,y,z from training data\n# This is our \"best guess\" for coordinates\nmean_x = train_lab['x_1'].mean()\nmean_y = train_lab['y_1'].mean()\nmean_z = train_lab['z_1'].mean()\n\nprint(f\"\\nAverage coordinates from training data:\")\nprint(f\"  x: {mean_x:.3f}\")\nprint(f\"  y: {mean_y:.3f}\")\nprint(f\"  z: {mean_z:.3f}\")\n\n# Build submission rows\nrows = []\n\nfor _, row in test_df.iterrows():\n    target_id = row['target_id']\n    sequence  = row['sequence']\n    \n    for i, letter in enumerate(sequence):\n        sub_row = {\n            'ID'     : f\"{target_id}_{i+1}\",\n            'resname': letter if letter in 'AUGC' else 'A',\n            'resid'  : i+1,\n        }\n        # 5 predictions with small random variation\n        for pred_num in range(1, 6):\n            noise = np.random.normal(0, 1.0, 3)\n            sub_row[f'x_{pred_num}'] = round(mean_x + noise[0], 3)\n            sub_row[f'y_{pred_num}'] = round(mean_y + noise[1], 3)\n            sub_row[f'z_{pred_num}'] = round(mean_z + noise[2], 3)\n        \n        rows.append(sub_row)\n\n# Save submission\nsubmission = pd.DataFrame(rows)\nsubmission.to_csv(\"/kaggle/working/submission.csv\", index=False)\n\nprint(f\"\\n Done!\")\nprint(f\"Total rows: {len(submission)}\")\nprint(f\"Sequences covered: {submission['ID'].str.rsplit('_',n=1).str[0].nunique()} / {len(test_df)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T21:37:36.147821Z","iopub.execute_input":"2026-03-24T21:37:36.148511Z","iopub.status.idle":"2026-03-24T21:37:48.016709Z","shell.execute_reply.started":"2026-03-24T21:37:36.148463Z","shell.execute_reply":"2026-03-24T21:37:48.015515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nweights_path = \"/kaggle/working/RhoFold/pretrained/RhoFold_pretrained.pt\"\n\nif not os.path.exists(weights_path):\n    !wget -q --show-progress \\\n        https://huggingface.co/cuhkaih/rhofold/resolve/main/rhofold_pretrained_params.pt \\\n        -O {weights_path}\n\nprint(\"Weights ready!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T21:37:53.832459Z","iopub.execute_input":"2026-03-24T21:37:53.833056Z","iopub.status.idle":"2026-03-24T21:37:53.840997Z","shell.execute_reply.started":"2026-03-24T21:37:53.832999Z","shell.execute_reply":"2026-03-24T21:37:53.839744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\nDATA_PATH = \"/kaggle/input/competitions/stanford-rna-3d-folding-2\"\ntest_df = pd.read_csv(f\"{DATA_PATH}/test_sequences.csv\")\n\nos.makedirs(\"/kaggle/working/fasta_inputs\", exist_ok=True)\nos.makedirs(\"/kaggle/working/predictions\", exist_ok=True)\n\nprint(f\"Loaded {len(test_df)} sequences\")\nprint(\"\\nSequences:\")\nfor _, row in test_df.iterrows():\n    print(f\"  {row['target_id']}: {len(row['sequence'])} nt\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T21:37:56.515375Z","iopub.execute_input":"2026-03-24T21:37:56.51583Z","iopub.status.idle":"2026-03-24T21:37:56.531412Z","shell.execute_reply.started":"2026-03-24T21:37:56.515792Z","shell.execute_reply":"2026-03-24T21:37:56.530255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"WEIGHTS  = \"/kaggle/working/RhoFold/pretrained/RhoFold_pretrained.pt\"\nMAX_LEN  = 100  \n# ↑ RhoFold+ reliably handles sequences up to ~1000 nt\n# anything longer gets fallback coordinates\n\nfor index, row in test_df.iterrows():\n    target_id = row['target_id']\n    sequence  = row['sequence']\n    seq_len   = len(sequence)\n    \n    output_dir = f\"/kaggle/working/predictions/{target_id}\"\n    pdb_file   = f\"{output_dir}/unrelaxed_model.pdb\"\n    \n    # Skip if already done\n    if os.path.exists(pdb_file):\n        print(f\"[{index+1}/28]   {target_id} already done\")\n        continue\n    \n    # Skip if too long for RhoFold+\n    if seq_len > MAX_LEN:\n        print(f\"[{index+1}/28]   {target_id} ({seq_len} nt) too long for RhoFold+ — will use fallback\")\n        continue\n    \n    print(f\"[{index+1}/28] Predicting {target_id} ({seq_len} nt)...\")\n    os.makedirs(output_dir, exist_ok=True)\n    \n    # Write FASTA file\n    fasta_path = f\"/kaggle/working/fasta_inputs/{target_id}.fasta\"\n    with open(fasta_path, 'w') as f:\n        f.write(f\">{target_id}\\n{sequence}\\n\")\n    \n    # Run RhoFold+ on CPU\n    os.system(\n        f\"python /kaggle/working/RhoFold/inference.py \"\n        f\"--input_fas {fasta_path} \"\n        f\"--single_seq_pred True \"\n        f\"--output_dir {output_dir} \"\n        f\"--ckpt {WEIGHTS} \"\n        f\"--device cpu \"\n        f\"--relax_steps 0\"\n    )\n    \n    if os.path.exists(pdb_file):\n        print(f\"  Success!\")\n    else:\n        print(f\"  Failed!\")\n\nprint(\"\\n All predictions attempted!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T21:37:59.805052Z","iopub.execute_input":"2026-03-24T21:37:59.805704Z","iopub.status.idle":"2026-03-24T21:43:15.112076Z","shell.execute_reply.started":"2026-03-24T21:37:59.805636Z","shell.execute_reply":"2026-03-24T21:43:15.110492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_coords(pdb_file):\n    coords = []\n    with open(pdb_file, 'r') as f:\n        for line in f:\n            if line.startswith(\"ATOM\") and \"C1'\" in line:\n                try:\n                    coords.append({\n                        'resname': line[17:20].strip(),\n                        'resid'  : int(line[22:26].strip()),\n                        'x': float(line[30:38].strip()),\n                        'y': float(line[38:46].strip()),\n                        'z': float(line[46:54].strip())\n                    })\n                except:\n                    continue\n    return coords\n\n# Load training data for fallback\ntrain_lab  = pd.read_csv(f\"{DATA_PATH}/train_labels.csv\", low_memory=False)\n# ↑ low_memory=False fixes the DtypeWarning from before\nfallback_x = train_lab['x_1'].mean()\nfallback_y = train_lab['y_1'].mean()\nfallback_z = train_lab['z_1'].mean()\n\nprint(f\"Fallback coords: ({fallback_x:.1f}, {fallback_y:.1f}, {fallback_z:.1f})\")\n\n# Build submission\nall_rows = []\n\nfor _, row in test_df.iterrows():\n    target_id = row['target_id']\n    sequence  = row['sequence']\n    pdb_file  = f\"/kaggle/working/predictions/{target_id}/unrelaxed_model.pdb\"\n    \n    if os.path.exists(pdb_file):\n        coords = extract_coords(pdb_file)\n        print(f\"✅ {target_id}: {len(coords)} residues from RhoFold+\")\n    else:\n        # Fallback for sequences RhoFold+ couldn't handle\n        coords = [\n            {\n                'resname': sequence[i] if sequence[i] in 'AUGC' else 'A',\n                'resid'  : i + 1,\n                'x': fallback_x,\n                'y': fallback_y,\n                'z': fallback_z\n            }\n            for i in range(len(sequence))\n        ]\n        print(f\"⚠️  {target_id}: fallback coords ({len(coords)} residues)\")\n    \n    for c in coords:\n        sub_row = {\n            'ID'     : f\"{target_id}_{c['resid']}\",\n            'resname': c['resname'],\n            'resid'  : c['resid'],\n        }\n        for pred_num in range(1, 6):\n            noise = np.random.normal(0, 0.1, 3)\n            sub_row[f'x_{pred_num}'] = round(c['x'] + noise[0], 3)\n            sub_row[f'y_{pred_num}'] = round(c['y'] + noise[1], 3)\n            sub_row[f'z_{pred_num}'] = round(c['z'] + noise[2], 3)\n        all_rows.append(sub_row)\n\nsubmission = pd.DataFrame(all_rows)\nsubmission.to_csv(\"/kaggle/working/submission.csv\", index=False)\n\nprint(f\"\\n Submission saved!\")\nprint(f\"Total rows:       {len(submission)}\")\nprint(f\"Sequences covered: {submission['ID'].str.rsplit('_',n=1).str[0].nunique()} / 28\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T21:44:32.437501Z","iopub.execute_input":"2026-03-24T21:44:32.438374Z","iopub.status.idle":"2026-03-24T21:44:49.605599Z","shell.execute_reply.started":"2026-03-24T21:44:32.438316Z","shell.execute_reply":"2026-03-24T21:44:49.604316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/working/submission.csv\")\n\nprint(\"========== FINAL CHECK ==========\")\nprint(f\"Total rows:       {len(df)}\")\nprint(f\"Missing values:   {df.isnull().sum().sum()}\")\nprint(f\"Sequences covered:{df['ID'].str.rsplit('_',n=1).str[0].nunique()} / 28\")\n\nprint(\"\\nPer sequence breakdown:\")\nfor tid in test_df['target_id']:\n    n = len(df[df['ID'].str.startswith(tid+'_')])\n    status = \"✅\" if n > 0 else \"❌ MISSING\"\n    print(f\"  {status}  {tid}: {n} rows\")\n\nprint(\"\\n=================================\")\nif df['ID'].str.rsplit('_',n=1).str[0].nunique() == 28 and df.isnull().sum().sum() == 0:\n    print(\" READY TO SUBMIT!\")\nelse:\n    print(\" ISSUES FOUND — check above\")\nprint(\"=================================\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T21:44:55.045837Z","iopub.execute_input":"2026-03-24T21:44:55.046939Z","iopub.status.idle":"2026-03-24T21:44:55.216108Z","shell.execute_reply.started":"2026-03-24T21:44:55.046889Z","shell.execute_reply":"2026-03-24T21:44:55.214885Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}