{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":118765,"databundleVersionId":15231210,"sourceType":"competition"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-24T14:37:59.666389Z","iopub.execute_input":"2026-01-24T14:37:59.666754Z","iopub.status.idle":"2026-01-24T14:38:20.825213Z","shell.execute_reply.started":"2026-01-24T14:37:59.66672Z","shell.execute_reply":"2026-01-24T14:38:20.823968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install biopython\n\nimport numpy as np\nimport pandas as pd\nimport random\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nfrom Bio import pairwise2\nfrom Bio.Seq import Seq\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T04:28:21.775747Z","iopub.execute_input":"2026-01-30T04:28:21.776546Z","iopub.status.idle":"2026-01-30T04:28:26.312853Z","shell.execute_reply.started":"2026-01-30T04:28:21.776511Z","shell.execute_reply":"2026-01-30T04:28:26.312177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_PATH = \"/kaggle/input/stanford-rna-3d-folding-2/\"\n\ntrain_seqs = pd.read_csv(DATA_PATH + \"train_sequences.csv\")\ntrain_labels = pd.read_csv(DATA_PATH + \"train_labels.csv\")\ntest_seqs = pd.read_csv(DATA_PATH + \"test_sequences.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T04:29:20.450061Z","iopub.execute_input":"2026-01-30T04:29:20.450386Z","iopub.status.idle":"2026-01-30T04:29:28.062902Z","shell.execute_reply.started":"2026-01-30T04:29:20.450356Z","shell.execute_reply":"2026-01-30T04:29:28.062314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    val_seqs = pd.read_csv(DATA_PATH + \"validation_sequences.csv\")\n    val_labels = pd.read_csv(DATA_PATH + \"validation_labels.csv\")\n\n    train_seqs = pd.concat([train_seqs, val_seqs], ignore_index=True)\n    train_labels = pd.concat([train_labels, val_labels], ignore_index=True)\n\n    print(\"Validation data added.\")\nexcept:\n    print(\"No validation data found.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T04:29:35.763358Z","iopub.execute_input":"2026-01-30T04:29:35.764174Z","iopub.status.idle":"2026-01-30T04:29:40.743464Z","shell.execute_reply.started":"2026-01-30T04:29:35.764135Z","shell.execute_reply":"2026-01-30T04:29:40.742498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_coord_dict(labels_df):\n    coord_dict = {}\n    for rid, group in labels_df.groupby(\n        labels_df[\"ID\"].apply(lambda x: x.rsplit(\"_\", 1)[0])\n    ):\n        group = group.sort_values(\"resid\")\n        coord_dict[rid] = group[[\"x_1\", \"y_1\", \"z_1\"]].values\n    return coord_dict\n\ncoord_dict = build_coord_dict(train_labels)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T04:29:45.569733Z","iopub.execute_input":"2026-01-30T04:29:45.570018Z","iopub.status.idle":"2026-01-30T04:29:59.34063Z","shell.execute_reply.started":"2026-01-30T04:29:45.569993Z","shell.execute_reply":"2026-01-30T04:29:59.339957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def find_similar_sequences(query_seq, train_df, coord_dict, top_n=5):\n    results = []\n    q = Seq(query_seq)\n\n    for _, row in train_df.iterrows():\n        tid = row[\"target_id\"]\n        tseq = row[\"sequence\"]\n\n        if tid not in coord_dict:\n            continue\n\n        if abs(len(tseq) - len(query_seq)) / max(len(tseq), len(query_seq)) > 0.4:\n            continue\n\n        aln = pairwise2.align.globalms(\n            q, tseq, 2, -1, -8, -0.3, one_alignment_only=True\n        )\n\n        if aln:\n            score = aln[0].score / (2 * min(len(q), len(tseq)))\n            results.append((tid, tseq, score, coord_dict[tid]))\n\n    results.sort(key=lambda x: x[2], reverse=True)\n    return results[:top_n]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T04:30:03.730305Z","iopub.execute_input":"2026-01-30T04:30:03.730953Z","iopub.status.idle":"2026-01-30T04:30:03.73671Z","shell.execute_reply.started":"2026-01-30T04:30:03.730927Z","shell.execute_reply":"2026-01-30T04:30:03.735937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def adapt_template(query_seq, template_seq, template_coords):\n    aln = pairwise2.align.globalms(\n        Seq(query_seq), Seq(template_seq), 2, -1, -8, -0.3, one_alignment_only=True\n    )\n\n    if not aln:\n        return np.zeros((len(query_seq), 3))\n\n    a_q, a_t = aln[0].seqA, aln[0].seqB\n    coords = np.full((len(query_seq), 3), np.nan)\n\n    qi, ti = 0, 0\n    for cq, ct in zip(a_q, a_t):\n        if cq != \"-\" and ct != \"-\":\n            if ti < len(template_coords):\n                coords[qi] = template_coords[ti]\n            qi += 1\n            ti += 1\n        elif cq != \"-\":\n            qi += 1\n        elif ct != \"-\":\n            ti += 1\n\n    # Fill missing coordinates\n    for i in range(len(coords)):\n        if np.isnan(coords[i, 0]):\n            if i > 0:\n                coords[i] = coords[i - 1] + [3, 0, 0]\n            else:\n                coords[i] = [i * 3, 0, 0]\n\n    return coords\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T04:30:30.306308Z","iopub.execute_input":"2026-01-30T04:30:30.306939Z","iopub.status.idle":"2026-01-30T04:30:30.313274Z","shell.execute_reply.started":"2026-01-30T04:30:30.306912Z","shell.execute_reply":"2026-01-30T04:30:30.312688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def refine_structure(coords, confidence):\n    refined = coords.copy()\n    strength = 0.7 * (1 - min(confidence, 0.95))\n    target_dist = 5.9\n\n    for i in range(len(refined) - 1):\n        dist = np.linalg.norm(refined[i + 1] - refined[i])\n        if dist < 5.8 or dist > 6.1:\n            direction = (refined[i + 1] - refined[i]) / (dist + 1e-6)\n            refined[i + 1] += direction * (target_dist - dist) * strength\n\n    return refined\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T04:30:33.745476Z","iopub.execute_input":"2026-01-30T04:30:33.746173Z","iopub.status.idle":"2026-01-30T04:30:33.751127Z","shell.execute_reply.started":"2026-01-30T04:30:33.746147Z","shell.execute_reply":"2026-01-30T04:30:33.75042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def generate_random_structure(sequence):\n    coords = np.zeros((len(sequence), 3))\n    for i in range(1, len(sequence)):\n        coords[i] = coords[i - 1] + [random.uniform(3.8, 4.2), 0, 0]\n    return coords\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T04:30:36.147416Z","iopub.execute_input":"2026-01-30T04:30:36.14795Z","iopub.status.idle":"2026-01-30T04:30:36.152132Z","shell.execute_reply.started":"2026-01-30T04:30:36.147924Z","shell.execute_reply":"2026-01-30T04:30:36.151504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_structures(sequence, train_df, coord_dict, n_preds=5):\n    preds = []\n    templates = find_similar_sequences(sequence, train_df, coord_dict, n_preds)\n\n    for tid, tseq, sim, tcoords in templates:\n        adapted = adapt_template(sequence, tseq, tcoords)\n        refined = refine_structure(adapted, sim)\n        noise = max(0.01, (0.4 - sim) * 0.1)\n        refined += np.random.normal(0, noise, refined.shape)\n        preds.append(refined)\n\n    while len(preds) < n_preds:\n        preds.append(generate_random_structure(sequence))\n\n    return preds\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T04:30:38.547453Z","iopub.execute_input":"2026-01-30T04:30:38.548168Z","iopub.status.idle":"2026-01-30T04:30:38.553976Z","shell.execute_reply.started":"2026-01-30T04:30:38.548139Z","shell.execute_reply":"2026-01-30T04:30:38.553061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_rna(coords, title=\"RNA Structure\"):\n    fig = plt.figure(figsize=(5,5))\n    ax = fig.add_subplot(111, projection=\"3d\")\n    ax.plot(coords[:,0], coords[:,1], coords[:,2], \"-o\", markersize=3)\n    ax.set_title(title)\n    plt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T04:30:46.292988Z","iopub.execute_input":"2026-01-30T04:30:46.293328Z","iopub.status.idle":"2026-01-30T04:30:46.297733Z","shell.execute_reply.started":"2026-01-30T04:30:46.293303Z","shell.execute_reply":"2026-01-30T04:30:46.297127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rows = []\n\nfor idx, row in test_seqs.iterrows():\n    if idx % 5 == 0:\n        print(f\"Processing {idx}/{len(test_seqs)}\")\n\n    preds = predict_structures(row[\"sequence\"], train_seqs, coord_dict)\n\n    for j, base in enumerate(row[\"sequence\"]):\n        r = {\n            \"ID\": f\"{row['target_id']}_{j+1}\",\n            \"resname\": base,\n            \"resid\": j + 1,\n        }\n        for i in range(5):\n            r[f\"x_{i+1}\"], r[f\"y_{i+1}\"], r[f\"z_{i+1}\"] = preds[i][j]\n        rows.append(r)\n\nsubmission = pd.DataFrame(rows)\n\ncols = [\"ID\", \"resname\", \"resid\"] + [\n    f\"{c}_{i}\" for i in range(1, 6) for c in [\"x\", \"y\", \"z\"]\n]\n\nsubmission[cols].to_csv(\"submission.csv\", index=False)\nprint(\"submission.csv generated!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-30T05:20:04.838275Z","iopub.execute_input":"2026-01-30T05:20:04.838499Z","iopub.status.idle":"2026-01-30T06:05:59.292719Z","shell.execute_reply.started":"2026-01-30T05:20:04.838479Z","shell.execute_reply":"2026-01-30T06:05:59.291986Z"}},"outputs":[],"execution_count":null}]}