{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210},{"sourceType":"datasetVersion","sourceId":8318191,"datasetId":4459124,"databundleVersionId":8449353},{"sourceType":"datasetVersion","sourceId":14452625,"datasetId":9231246,"databundleVersionId":15272827},{"sourceType":"datasetVersion","sourceId":7639698,"datasetId":4299272,"databundleVersionId":7736182}],"dockerImageVersionId":31286,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport re\nimport numpy as np\nimport pandas as pd\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.metrics.pairwise import cosine_similarity","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-17T16:20:41.638366Z","iopub.execute_input":"2026-03-17T16:20:41.638693Z","iopub.status.idle":"2026-03-17T16:20:42.888483Z","shell.execute_reply.started":"2026-03-17T16:20:41.638668Z","shell.execute_reply":"2026-03-17T16:20:42.887595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_SEQ_CSV    = \"/kaggle/input/competitions/stanford-rna-3d-folding-2/train_sequences.csv\"\nTRAIN_LABELS_CSV = \"/kaggle/input/competitions/stanford-rna-3d-folding-2/train_labels.csv\"\nTEST_SEQ_CSV     = \"/kaggle/input/competitions/stanford-rna-3d-folding-2/test_sequences.csv\"\n\ntrain_seq    = pd.read_csv(TRAIN_SEQ_CSV)\ntrain_labels = pd.read_csv(TRAIN_LABELS_CSV)\ntest_seq     = pd.read_csv(TEST_SEQ_CSV)\n\nprint(\"train_seq:\", train_seq.shape)\nprint(\"train_labels:\", train_labels.shape)\nprint(\"test_seq:\", test_seq.shape)\nprint(train_labels.columns.tolist())\nprint(train_labels.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T16:20:42.889881Z","iopub.execute_input":"2026-03-17T16:20:42.890347Z","iopub.status.idle":"2026-03-17T16:20:53.727132Z","shell.execute_reply.started":"2026-03-17T16:20:42.890316Z","shell.execute_reply":"2026-03-17T16:20:53.726277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def guess_col(df, candidates):\n    for c in candidates:\n        if c in df.columns:\n            return c\n    raise ValueError(f\"Cannot find any of {candidates} in {list(df.columns)}\")\n\n# Automatic column identification\ntrain_target_col = guess_col(train_seq, [\"target_id\", \"id\", \"ID\"])\ntrain_seq_col    = guess_col(train_seq, [\"sequence\", \"seq\"])\ntest_target_col  = guess_col(test_seq,  [\"target_id\", \"id\", \"ID\"])\ntest_seq_col     = guess_col(test_seq,  [\"sequence\", \"seq\"])\nlabel_id_col     = guess_col(train_labels, [\"ID\", \"id\"])\n\nprint(train_target_col, train_seq_col, test_target_col, test_seq_col, label_id_col)\n\ndef extract_target_from_label_id(x):\n    \"\"\"Extract 'targetid' from 'targetid_123'\"\"\"\n    m = re.match(r\"^(.*)_([0-9]+)$\", str(x))\n    return m.group(1) if m else str(x)\n\ndef resample_coords(coords, new_len):\n    \"\"\"Linear interpolation to adapt template coordinate length to test length\"\"\"\n    coords = np.asarray(coords, dtype=float)\n    old_len = len(coords)\n    if old_len == 0:\n        return np.zeros((new_len, 3), dtype=float)\n    if old_len == 1:\n        return np.repeat(coords, new_len, axis=0)\n    if old_len == new_len:\n        return coords.copy()\n    \n    old_x = np.linspace(0, 1, old_len)\n    new_x = np.linspace(0, 1, new_len)\n    new_coords = np.zeros((new_len, 3), dtype=float)\n    for j in range(3):\n        new_coords[:, j] = np.interp(new_x, old_x, coords[:, j])\n    return new_coords\n\ndef add_perturbation(coords, scale, seed):\n    \"\"\"Add small perturbations to generate different conformers\"\"\"\n    rng = np.random.default_rng(seed)\n    return coords + rng.normal(0.0, scale, coords.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T16:20:53.728267Z","iopub.execute_input":"2026-03-17T16:20:53.728587Z","iopub.status.idle":"2026-03-17T16:20:53.741482Z","shell.execute_reply.started":"2026-03-17T16:20:53.728558Z","shell.execute_reply":"2026-03-17T16:20:53.740597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels = train_labels.copy()\ntrain_labels[\"template_target\"] = train_labels[label_id_col].apply(extract_target_from_label_id)\n\n# Check if coordinate columns exist\nfor c in [\"x_1\", \"y_1\", \"z_1\"]:\n    if c not in train_labels.columns:\n        raise ValueError(f\"Missing column: {c}\")\n\ntemplate_library = {}\nfor tid, g in train_labels.groupby(\"template_target\"):\n    if \"resid\" in g.columns:\n        g = g.sort_values(\"resid\")\n    coords = g[[\"x_1\", \"y_1\", \"z_1\"]].to_numpy(dtype=float)\n    # Handle NaN values\n    coords = np.nan_to_num(coords, nan=0.0, posinf=0.0, neginf=0.0)\n    template_library[tid] = coords\n\nprint(f\"Template library size: {len(template_library)}\")\nbad_count = sum(np.isnan(v).any() for v in template_library.values())\nprint(f\"Templates still containing NaN: {bad_count}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T16:20:53.743622Z","iopub.execute_input":"2026-03-17T16:20:53.744122Z","iopub.status.idle":"2026-03-17T16:21:09.427153Z","shell.execute_reply.started":"2026-03-17T16:20:53.744093Z","shell.execute_reply":"2026-03-17T16:21:09.426343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seq_to_kmer_str(seq, k=3):\n    seq = str(seq)\n    return \" \".join([seq[i:i+k] for i in range(max(1, len(seq)-k+1))])\n\n# Use both train and validation sequences for the matching pool, but only use coordinates from train\nVAL_SEQ_CSV = \"/kaggle/input/competitions/stanford-rna-3d-folding-2/validation_sequences.csv\"\nval_seq = pd.read_csv(VAL_SEQ_CSV)\n\n# Merge sequences (CSVs are small)\nall_seq = pd.concat([train_seq, val_seq], ignore_index=True)\n\n# Keep only records that exist in the template_library with coordinates\ntrain_records = all_seq[\n    all_seq[train_target_col].astype(str).isin(template_library.keys())\n].reset_index(drop=True)\n\nprint(f\"Valid template records: {len(train_records)}\")\n\n# Vectorized matching logic remains unchanged\ntrain_kmer = train_records[train_seq_col].apply(seq_to_kmer_str).tolist()\ntest_kmer  = test_seq[test_seq_col].apply(seq_to_kmer_str).tolist()\n\nvectorizer = TfidfVectorizer()\ntfidf_all  = vectorizer.fit_transform(train_kmer + test_kmer)\n\ntrain_vecs = tfidf_all[:len(train_kmer)]\ntest_vecs  = tfidf_all[len(train_kmer):]\n\nsim_matrix = cosine_similarity(test_vecs, train_vecs)\n\nbest_template_ids = [\n    str(train_records.iloc[sim_matrix[i].argmax()][train_target_col])\n    for i in range(len(test_seq))\n]\n\nprint(\"Example match:\", test_seq.iloc[0][test_target_col], \"->\", best_template_ids[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T16:21:09.42832Z","iopub.execute_input":"2026-03-17T16:21:09.428658Z","iopub.status.idle":"2026-03-17T16:21:13.894877Z","shell.execute_reply.started":"2026-03-17T16:21:09.428618Z","shell.execute_reply":"2026-03-17T16:21:13.894095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def simple_align_and_map(test_seq, template_seq, template_coords):\n    \"\"\"Map coordinates based on sequence alignment, more robust than linear interpolation\"\"\"\n    n_test = len(test_seq)\n    n_tmpl = len(template_coords)\n    coords_out = np.zeros((n_test, 3), dtype=float)\n    tmpl_pos = 0\n    for i in range(n_test):\n        best_pos = min(tmpl_pos, n_tmpl - 1)\n        coords_out[i] = template_coords[best_pos]\n        if tmpl_pos < n_tmpl - 1:\n            tmpl_pos += 1\n    return coords_out\n\n# Retrieve top 10 templates for each test sequence to ensure 5 valid conformers\ntop10_template_ids = [\n    [str(train_records.iloc[j][train_target_col])\n     for j in sim_matrix[i].argsort()[-10:][::-1]]\n    for i in range(len(test_seq))\n]\n\nrows = []\n\nfor idx in range(len(test_seq)):\n    row       = test_seq.iloc[idx]\n    target_id = str(row[test_target_col])\n    sequence  = str(row[test_seq_col])\n    seq_len   = len(sequence)\n\n    # Generate 5 conformers using 5 different templates\n    confs = []\n    # Use the top 5 unique templates found in the matching pool\n    for rank in range(5):\n        tid = top10_template_ids[idx][rank]\n        tmpl_coords = template_library.get(tid, np.zeros((1, 3)))\n        \n        # Get template sequence\n        tmpl_seq_rows = train_records[train_records[train_target_col] == tid]\n        if len(tmpl_seq_rows) > 0:\n            tmpl_seq = str(tmpl_seq_rows.iloc[0][train_seq_col])\n            conf = simple_align_and_map(sequence, tmpl_seq, tmpl_coords)\n        else:\n            # Fallback to resampling if sequence row is missing\n            conf = resample_coords(tmpl_coords, seq_len)\n        confs.append(conf)\n\n    # Construct the output rows for each residue\n    for i, resname in enumerate(sequence, start=1):\n        out = {\"ID\": f\"{target_id}_{i}\", \"resname\": resname, \"resid\": i}\n        for k in range(5):\n            xyz = confs[k][i-1]\n            out[f\"x_{k+1}\"] = float(xyz[0])\n            out[f\"y_{k+1}\"] = float(xyz[1])\n            out[f\"z_{k+1}\"] = float(xyz[2])\n        rows.append(out)\n\nsubmission = pd.DataFrame(rows)\nprint(\"Submission shape:\", submission.shape)\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T16:21:13.89589Z","iopub.execute_input":"2026-03-17T16:21:13.896196Z","iopub.status.idle":"2026-03-17T16:21:14.07579Z","shell.execute_reply.started":"2026-03-17T16:21:13.896167Z","shell.execute_reply":"2026-03-17T16:21:14.075051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols = [c for c in submission.columns if c not in [\"ID\", \"resname\"]]\n\nsubmission[num_cols] = submission[num_cols].replace([np.inf, -np.inf], np.nan)\nsubmission[num_cols] = submission[num_cols].fillna(0.0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T16:21:14.076748Z","iopub.execute_input":"2026-03-17T16:21:14.077019Z","iopub.status.idle":"2026-03-17T16:21:14.090585Z","shell.execute_reply.started":"2026-03-17T16:21:14.076989Z","shell.execute_reply":"2026-03-17T16:21:14.089701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"any NaN after fix:\", submission.isna().any().any())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T16:21:14.091555Z","iopub.execute_input":"2026-03-17T16:21:14.091843Z","iopub.status.idle":"2026-03-17T16:21:14.100033Z","shell.execute_reply.started":"2026-03-17T16:21:14.091819Z","shell.execute_reply":"2026-03-17T16:21:14.099021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"expected_cols = (\n    [\"ID\", \"resname\", \"resid\"] +\n    [f\"{a}_{k}\" for k in range(1,6) for a in [\"x\",\"y\",\"z\"]]\n)\n\nsubmission = submission[expected_cols]\nsubmission.to_csv(\"/kaggle/working/submission.csv\", index=False)\nprint(\"saved:\", os.path.exists(\"/kaggle/working/submission.csv\"))\nprint(pd.read_csv(\"/kaggle/working/submission.csv\").head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T16:21:14.101248Z","iopub.execute_input":"2026-03-17T16:21:14.101582Z","iopub.status.idle":"2026-03-17T16:21:14.419909Z","shell.execute_reply.started":"2026-03-17T16:21:14.101548Z","shell.execute_reply":"2026-03-17T16:21:14.419026Z"}},"outputs":[],"execution_count":null}]}