{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210,"isSourceIdPinned":false}],"dockerImageVersionId":31286,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-17T15:55:42.660005Z","iopub.execute_input":"2026-03-17T15:55:42.660439Z","iopub.status.idle":"2026-03-17T15:55:59.275576Z","shell.execute_reply.started":"2026-03-17T15:55:42.660398Z","shell.execute_reply":"2026-03-17T15:55:59.274298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Load data ─────────────────────────────────────────────\ntest_seq  = pd.read_csv('/kaggle/input/competitions/stanford-rna-3d-folding-2/test_sequences.csv')\ntrain_seq = pd.read_csv('/kaggle/input/competitions/stanford-rna-3d-folding-2/train_sequences.csv')\ntrain_lab = pd.read_csv('/kaggle/input/competitions/stanford-rna-3d-folding-2/train_labels.csv', low_memory=False)\n\nprint(f\"Test: {len(test_seq)} sequences\")\nprint(f\"Train: {len(train_seq)} sequences\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T15:57:44.362503Z","iopub.execute_input":"2026-03-17T15:57:44.363027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Sequence similarity: k-mer overlap ───────────────────\ndef kmer_similarity(seq1, seq2, k=3):\n    \"\"\"Compute Jaccard similarity between k-mer sets of two sequences\"\"\"\n    kmers1 = set(seq1[i:i+k] for i in range(len(seq1) - k + 1))\n    kmers2 = set(seq2[i:i+k] for i in range(len(seq2) - k + 1))\n    if not kmers1 or not kmers2:\n        return 0.0\n    intersection = len(kmers1 & kmers2)\n    union = len(kmers1 | kmers2)\n    return intersection / union\n\ndef find_best_template(test_seq_str, train_df, top_n=1):\n    \"\"\"Find most similar training sequence using k-mer similarity + length penalty\"\"\"\n    test_len = len(test_seq_str)\n    best_id, best_score = None, -1\n\n    for _, row in train_df.iterrows():\n        train_seq_str = row['sequence']\n        train_len = len(train_seq_str)\n\n        # Length penalty: penalize if length ratio is too different\n        len_ratio = min(test_len, train_len) / max(test_len, train_len)\n        if len_ratio < 0.3:  # skip sequences with very different lengths\n            continue\n\n        sim = kmer_similarity(test_seq_str, train_seq_str, k=3)\n        score = sim * len_ratio  # combined score\n\n        if score > best_score:\n            best_score = score\n            best_id = row['target_id']\n\n    return best_id, best_score\n\n# ── Build coordinate lookup from train_labels ─────────────\nprint(\"Building coordinate lookup...\")\ntrain_lab_clean = train_lab.dropna(subset=['x_1', 'y_1', 'z_1'])\ncoord_lookup = {}  # target_id -> DataFrame of coords sorted by resid\nfor tid, group in train_lab_clean.groupby(train_lab_clean['ID'].str.extract(r'^(.+)_\\d+$')[0]):\n    coord_lookup[tid] = group.sort_values('resid').reset_index(drop=True)\nprint(f\"Loaded {len(coord_lookup)} templates\")\n\n# ── Helper: scale coordinates to match test sequence length ──\ndef get_coords_for_length(ref_coords, target_len):\n    \"\"\"\n    Map reference coordinates onto target length.\n    If ref is longer: subsample evenly.\n    If ref is shorter: repeat last residue coords.\n    \"\"\"\n    ref_len = len(ref_coords)\n    coords = np.zeros((target_len, 3))\n\n    if ref_len >= target_len:\n        # Subsample: pick evenly spaced indices\n        indices = np.round(np.linspace(0, ref_len - 1, target_len)).astype(int)\n        for i, idx in enumerate(indices):\n            coords[i] = [ref_coords.iloc[idx]['x_1'],\n                         ref_coords.iloc[idx]['y_1'],\n                         ref_coords.iloc[idx]['z_1']]\n    else:\n        # Copy what we have, then extend by repeating last coord\n        for i in range(ref_len):\n            coords[i] = [ref_coords.iloc[i]['x_1'],\n                         ref_coords.iloc[i]['y_1'],\n                         ref_coords.iloc[i]['z_1']]\n        for i in range(ref_len, target_len):\n            coords[i] = coords[ref_len - 1]  # repeat last\n\n    return coords","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T16:04:05.860305Z","iopub.execute_input":"2026-03-17T16:04:05.861076Z","iopub.status.idle":"2026-03-17T16:04:20.679744Z","shell.execute_reply.started":"2026-03-17T16:04:05.861041Z","shell.execute_reply":"2026-03-17T16:04:20.678815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Main loop ─────────────────────────────────────────────\nrows = []\nnp.random.seed(42)\n\n# Fallback coordinate stats (for when no template found)\nx_mean, x_std = train_lab_clean['x_1'].mean(), train_lab_clean['x_1'].std()\ny_mean, y_std = train_lab_clean['y_1'].mean(), train_lab_clean['y_1'].std()\nz_mean, z_std = train_lab_clean['z_1'].mean(), train_lab_clean['z_1'].std()\n\nfor _, test_row in test_seq.iterrows():\n    target_id = test_row['target_id']\n    sequence  = test_row['sequence']\n    seq_len   = len(sequence)\n\n    # Find best matching template\n    best_id, best_score = find_best_template(sequence, train_seq)\n    print(f\"{target_id} (len={seq_len}) → template: {best_id} (score={best_score:.3f})\")\n\n    # Get base coordinates from template\n    if best_id and best_id in coord_lookup and best_score > 0.05:\n        ref_coords = coord_lookup[best_id]\n        base_xyz = get_coords_for_length(ref_coords, seq_len)\n        use_template = True\n    else:\n        # No good template: use random from training distribution\n        base_xyz = np.column_stack([\n            np.random.normal(x_mean, x_std, seq_len),\n            np.random.normal(y_mean, y_std, seq_len),\n            np.random.normal(z_mean, z_std, seq_len),\n        ])\n        use_template = False\n\n    for resid in range(1, seq_len + 1):\n        resname = sequence[resid - 1]\n        entry = {\n            'ID':      f'{target_id}_{resid}',\n            'resname': resname,\n            'resid':   resid,\n        }\n        bx, by, bz = base_xyz[resid - 1]\n\n        # 5 predictions: pred 1 = template as-is, preds 2-5 = increasing noise\n        entry['x_1'], entry['y_1'], entry['z_1'] = bx, by, bz\n        for k in range(2, 6):\n            noise = 3.0 * (k - 1)\n            entry[f'x_{k}'] = bx + np.random.normal(0, noise)\n            entry[f'y_{k}'] = by + np.random.normal(0, noise)\n            entry[f'z_{k}'] = bz + np.random.normal(0, noise)\n\n        rows.append(entry)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T16:04:42.879291Z","iopub.execute_input":"2026-03-17T16:04:42.879631Z","iopub.status.idle":"2026-03-17T16:04:57.513074Z","shell.execute_reply.started":"2026-03-17T16:04:42.879606Z","shell.execute_reply":"2026-03-17T16:04:57.51214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Save ──────────────────────────────────────────────────\nsubmission = pd.DataFrame(rows)\nprint(f\"\\nShape: {submission.shape}\")\nprint(f\"NaN count: {submission.isna().sum().sum()}\")\nprint(submission.head(3).to_string())\n\nsubmission.to_csv('/kaggle/working/submission.csv', index=False)\nprint(\"\\n✓ submission.csv saved!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-17T16:05:34.744978Z","iopub.execute_input":"2026-03-17T16:05:34.74532Z","iopub.status.idle":"2026-03-17T16:05:35.059306Z","shell.execute_reply.started":"2026-03-17T16:05:34.745293Z","shell.execute_reply":"2026-03-17T16:05:35.058468Z"}},"outputs":[],"execution_count":null}]}