{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210}],"dockerImageVersionId":31328,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:05:55.885475Z","iopub.execute_input":"2026-03-24T20:05:55.885858Z","iopub.status.idle":"2026-03-24T20:06:25.249562Z","shell.execute_reply.started":"2026-03-24T20:05:55.885796Z","shell.execute_reply":"2026-03-24T20:06:25.248117Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport math\nimport numpy as np\nimport pandas as pd\nfrom difflib import SequenceMatcher\nfrom collections import Counter","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:06:33.656546Z","iopub.execute_input":"2026-03-24T20:06:33.657038Z","iopub.status.idle":"2026-03-24T20:06:33.662979Z","shell.execute_reply.started":"2026-03-24T20:06:33.657003Z","shell.execute_reply":"2026-03-24T20:06:33.661894Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Load data","metadata":{}},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/competitions/stanford-rna-3d-folding-2\"\n\nsample_path = os.path.join(DATA_DIR, \"sample_submission.csv\")\ntest_path = os.path.join(DATA_DIR, \"test_sequences.csv\")\ntrain_seq_path = os.path.join(DATA_DIR, \"train_sequences.csv\")\ntrain_label_path = os.path.join(DATA_DIR, \"train_labels.csv\")\nval_seq_path = os.path.join(DATA_DIR, \"validation_sequences.csv\")\nval_label_path = os.path.join(DATA_DIR, \"validation_labels.csv\")\n\nsample_sub = pd.read_csv(sample_path)\ntest_seq = pd.read_csv(test_path)\ntrain_seq = pd.read_csv(train_seq_path)\ntrain_lab = pd.read_csv(train_label_path, low_memory=False)\nval_seq = pd.read_csv(val_seq_path)\nval_lab = pd.read_csv(val_label_path, low_memory=False)\n\nprint(\"sample_submission shape:\", sample_sub.shape)\nprint(\"test_sequences shape:\", test_seq.shape)\nprint(\"train_sequences shape:\", train_seq.shape)\nprint(\"train_labels shape:\", train_lab.shape)\nprint(\"validation_sequences shape:\", val_seq.shape)\nprint(\"validation_labels shape:\", val_lab.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:07:09.061804Z","iopub.execute_input":"2026-03-24T20:07:09.062315Z","iopub.status.idle":"2026-03-24T20:07:24.559317Z","shell.execute_reply.started":"2026-03-24T20:07:09.06228Z","shell.execute_reply":"2026-03-24T20:07:24.557893Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Inspect the competition format","metadata":{}},{"cell_type":"code","source":"print(sample_sub.columns.tolist())\ndisplay(sample_sub.head())\ndisplay(test_seq.head())\ndisplay(train_seq.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:07:42.202669Z","iopub.execute_input":"2026-03-24T20:07:42.203061Z","iopub.status.idle":"2026-03-24T20:07:42.259144Z","shell.execute_reply.started":"2026-03-24T20:07:42.203033Z","shell.execute_reply":"2026-03-24T20:07:42.2577Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Hhelper functions for parsing and sequence features","metadata":{}},{"cell_type":"code","source":"def extract_target_id(full_id: str) -> str:\n    return str(full_id).rsplit(\"_\", 1)[0]\n\ndef normalize_sequence(seq: str) -> str:\n    seq = str(seq).upper()\n    seq = re.sub(r\"[^ACGU]\", \"\", seq)\n    return seq\n\nKMER_LIST = [a + b for a in \"ACGU\" for b in \"ACGU\"]\nKMER_INDEX = {k: i for i, k in enumerate(KMER_LIST)}\n\ndef seq_features(seq: str):\n    seq = normalize_sequence(seq)\n    n = len(seq)\n\n    comp = np.zeros(4, dtype=float)\n    for i, nt in enumerate(\"ACGU\"):\n        comp[i] = seq.count(nt) / n if n > 0 else 0.0\n\n    kmer = np.zeros(len(KMER_LIST), dtype=float)\n    if n >= 2:\n        counts = Counter(seq[i:i+2] for i in range(n - 1))\n        total = sum(counts.values())\n        for k, v in counts.items():\n            if k in KMER_INDEX:\n                kmer[KMER_INDEX[k]] = v / total\n\n    return {\n        \"seq\": seq,\n        \"length\": n,\n        \"comp\": comp,\n        \"kmer\": kmer\n    }\n\ndef sequence_match_ratio(seq1: str, seq2: str) -> float:\n    return SequenceMatcher(None, seq1, seq2).ratio()\n\ndef length_similarity(n1: int, n2: int) -> float:\n    if max(n1, n2) == 0:\n        return 0.0\n    return 1.0 - abs(n1 - n2) / max(n1, n2)\n\ndef composition_similarity(comp1, comp2) -> float:\n    return 1.0 - np.mean(np.abs(comp1 - comp2))\n\ndef kmer_similarity(k1, k2) -> float:\n    # cosine similarity\n    denom = np.linalg.norm(k1) * np.linalg.norm(k2)\n    if denom == 0:\n        return 0.0\n    return float(np.dot(k1, k2) / denom)\n\ndef combined_similarity(fq, ft, weights):\n    s_match = sequence_match_ratio(fq[\"seq\"], ft[\"seq\"])\n    s_len = length_similarity(fq[\"length\"], ft[\"length\"])\n    s_comp = composition_similarity(fq[\"comp\"], ft[\"comp\"])\n    s_kmer = kmer_similarity(fq[\"kmer\"], ft[\"kmer\"])\n\n    score = (\n        weights[\"match\"] * s_match +\n        weights[\"length\"] * s_len +\n        weights[\"comp\"] * s_comp +\n        weights[\"kmer\"] * s_kmer\n    )\n    return float(score)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:08:18.24212Z","iopub.execute_input":"2026-03-24T20:08:18.242434Z","iopub.status.idle":"2026-03-24T20:08:18.256106Z","shell.execute_reply.started":"2026-03-24T20:08:18.242405Z","shell.execute_reply":"2026-03-24T20:08:18.255234Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Build training prototypes from `x_1, y_1, z_1`","metadata":{}},{"cell_type":"code","source":"train_lab = train_lab.copy()\ntrain_lab[\"target_id\"] = train_lab[\"ID\"].astype(str).apply(extract_target_id)\n\ntrain_proto = train_lab[[\"target_id\", \"resname\", \"resid\", \"x_1\", \"y_1\", \"z_1\"]].copy()\ntrain_proto = train_proto.dropna(subset=[\"x_1\", \"y_1\", \"z_1\"])\n\nprototype_dict = {}\n\nfor target_id, g in train_proto.groupby(\"target_id\", sort=False):\n    g = g.sort_values(\"resid\").reset_index(drop=True)\n\n    coords = g[[\"x_1\", \"y_1\", \"z_1\"]].to_numpy(dtype=float)\n    resnames = g[\"resname\"].astype(str).tolist()\n\n    prototype_dict[target_id] = {\n        \"coords\": coords,\n        \"resname\": resnames,\n        \"length\": len(coords)\n    }\n\nprint(\"Number of training targets with coordinate prototypes:\", len(prototype_dict))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:09:05.622184Z","iopub.execute_input":"2026-03-24T20:09:05.623023Z","iopub.status.idle":"2026-03-24T20:09:17.815988Z","shell.execute_reply.started":"2026-03-24T20:09:05.622987Z","shell.execute_reply":"2026-03-24T20:09:17.814884Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Keep only training sequences that have structures","metadata":{}},{"cell_type":"code","source":"train_seq = train_seq.copy()\ntrain_seq[\"sequence\"] = train_seq[\"sequence\"].astype(str).apply(normalize_sequence)\n\ntrain_seq_avail = train_seq[train_seq[\"target_id\"].isin(prototype_dict.keys())].copy().reset_index(drop=True)\n\nprint(\"Training sequences available for template matching:\", train_seq_avail.shape[0])\ndisplay(train_seq_avail.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:11:01.487702Z","iopub.execute_input":"2026-03-24T20:11:01.48852Z","iopub.status.idle":"2026-03-24T20:11:01.603397Z","shell.execute_reply.started":"2026-03-24T20:11:01.488486Z","shell.execute_reply":"2026-03-24T20:11:01.602476Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Precompute training sequence features","metadata":{}},{"cell_type":"code","source":"train_features = {}\n\nfor _, row in train_seq_avail.iterrows():\n    tid = row[\"target_id\"]\n    seq = row[\"sequence\"]\n    train_features[tid] = seq_features(seq)\n\nprint(\"Precomputed training feature vectors:\", len(train_features))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:11:27.767282Z","iopub.execute_input":"2026-03-24T20:11:27.767652Z","iopub.status.idle":"2026-03-24T20:11:30.351662Z","shell.execute_reply.started":"2026-03-24T20:11:27.767623Z","shell.execute_reply":"2026-03-24T20:11:30.350219Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Interpolation and small geometry helpers","metadata":{}},{"cell_type":"code","source":"def interpolate_coords(coords, new_len):\n    old_len = coords.shape[0]\n\n    if old_len == new_len:\n        return coords.copy()\n\n    old_idx = np.linspace(0, 1, old_len)\n    new_idx = np.linspace(0, 1, new_len)\n\n    out = np.zeros((new_len, 3), dtype=float)\n    for j in range(3):\n        out[:, j] = np.interp(new_idx, old_idx, coords[:, j])\n\n    return out\n\ndef center_coords(coords):\n    return coords - coords.mean(axis=0, keepdims=True)\n\ndef scale_coords(coords):\n    c = center_coords(coords)\n    rms = np.sqrt(np.mean(np.sum(c**2, axis=1)))\n    if rms < 1e-8:\n        return c\n    return c / rms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:11:56.35714Z","iopub.execute_input":"2026-03-24T20:11:56.357735Z","iopub.status.idle":"2026-03-24T20:11:56.367214Z","shell.execute_reply.started":"2026-03-24T20:11:56.357689Z","shell.execute_reply":"2026-03-24T20:11:56.365543Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Kabsch alignment and TM-like validation metric","metadata":{}},{"cell_type":"code","source":"def kabsch_align(P, Q):\n    \"\"\"\n    Align P onto Q.\n    P, Q: arrays of shape (n, 3)\n    Returns aligned P.\n    \"\"\"\n    Pc = P - P.mean(axis=0, keepdims=True)\n    Qc = Q - Q.mean(axis=0, keepdims=True)\n\n    H = Pc.T @ Qc\n    U, S, Vt = np.linalg.svd(H)\n    R = Vt.T @ U.T\n\n    if np.linalg.det(R) < 0:\n        Vt[-1, :] *= -1\n        R = Vt.T @ U.T\n\n    P_aligned = Pc @ R + Q.mean(axis=0, keepdims=True)\n    return P_aligned\n\ndef tm_like_score(pred, truth):\n    \"\"\"\n    Approximate TM-score after Kabsch alignment.\n    \"\"\"\n    pred = np.asarray(pred, dtype=float)\n    truth = np.asarray(truth, dtype=float)\n\n    n = min(len(pred), len(truth))\n    if n == 0:\n        return 0.0\n\n    pred = pred[:n]\n    truth = truth[:n]\n\n    pred = kabsch_align(pred, truth)\n\n    d = np.sqrt(np.sum((pred - truth) ** 2, axis=1))\n\n    # standard-ish d0 scaling\n    d0 = max(0.5, 1.24 * ((max(n, 16) - 15) ** (1/3)) - 1.8)\n    score = np.mean(1.0 / (1.0 + (d / d0) ** 2))\n    return float(score)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:12:28.636702Z","iopub.execute_input":"2026-03-24T20:12:28.637485Z","iopub.status.idle":"2026-03-24T20:12:28.646934Z","shell.execute_reply.started":"2026-03-24T20:12:28.637449Z","shell.execute_reply":"2026-03-24T20:12:28.645829Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Build validation truth dictionary","metadata":{}},{"cell_type":"code","source":"val_lab = val_lab.copy()\nval_lab[\"target_id\"] = val_lab[\"ID\"].astype(str).apply(extract_target_id)\nval_seq = val_seq.copy()\nval_seq[\"sequence\"] = val_seq[\"sequence\"].astype(str).apply(normalize_sequence)\n\nval_truth_dict = {}\n\nval_lab_first = val_lab[[\"target_id\", \"resname\", \"resid\", \"x_1\", \"y_1\", \"z_1\"]].copy()\nval_lab_first = val_lab_first.dropna(subset=[\"x_1\", \"y_1\", \"z_1\"])\n\nfor target_id, g in val_lab_first.groupby(\"target_id\", sort=False):\n    g = g.sort_values(\"resid\").reset_index(drop=True)\n    coords = g[[\"x_1\", \"y_1\", \"z_1\"]].to_numpy(dtype=float)\n    val_truth_dict[target_id] = coords\n\nprint(\"Validation targets with truth coordinates:\", len(val_truth_dict))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:12:53.881871Z","iopub.execute_input":"2026-03-24T20:12:53.882267Z","iopub.status.idle":"2026-03-24T20:12:53.93113Z","shell.execute_reply.started":"2026-03-24T20:12:53.882238Z","shell.execute_reply":"2026-03-24T20:12:53.929695Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Candidate weight settings to tune on validation","metadata":{}},{"cell_type":"code","source":"weight_grid = [\n    {\"match\": 0.55, \"length\": 0.15, \"comp\": 0.10, \"kmer\": 0.20},\n    {\"match\": 0.60, \"length\": 0.10, \"comp\": 0.10, \"kmer\": 0.20},\n    {\"match\": 0.50, \"length\": 0.15, \"comp\": 0.05, \"kmer\": 0.30},\n    {\"match\": 0.45, \"length\": 0.15, \"comp\": 0.10, \"kmer\": 0.30},\n    {\"match\": 0.65, \"length\": 0.10, \"comp\": 0.05, \"kmer\": 0.20},\n    {\"match\": 0.50, \"length\": 0.20, \"comp\": 0.10, \"kmer\": 0.20},\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:13:17.23165Z","iopub.execute_input":"2026-03-24T20:13:17.231992Z","iopub.status.idle":"2026-03-24T20:13:17.238094Z","shell.execute_reply.started":"2026-03-24T20:13:17.231963Z","shell.execute_reply":"2026-03-24T20:13:17.236872Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Function to rank training templates for a query sequence","metadata":{}},{"cell_type":"code","source":"def rank_training_templates(query_seq, weights, top_k=5):\n    fq = seq_features(query_seq)\n\n    rows = []\n    for tid, ft in train_features.items():\n        score = combined_similarity(fq, ft, weights)\n        rows.append((tid, score))\n\n    rows = sorted(rows, key=lambda x: x[1], reverse=True)\n    return rows[:top_k]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:13:37.028849Z","iopub.execute_input":"2026-03-24T20:13:37.029695Z","iopub.status.idle":"2026-03-24T20:13:37.035272Z","shell.execute_reply.started":"2026-03-24T20:13:37.029661Z","shell.execute_reply":"2026-03-24T20:13:37.034212Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Tune weights on validation","metadata":{}},{"cell_type":"code","source":"tuning_rows = []\n\nfor weights in weight_grid:\n    target_scores = []\n\n    for _, row in val_seq.iterrows():\n        target_id = row[\"target_id\"]\n        seq = row[\"sequence\"]\n\n        if target_id not in val_truth_dict:\n            continue\n\n        ranked = rank_training_templates(seq, weights, top_k=5)\n\n        truth_coords = val_truth_dict[target_id]\n        best_local = -np.inf\n\n        for matched_tid, sim_score in ranked:\n            proto_coords = prototype_dict[matched_tid][\"coords\"]\n            pred_coords = interpolate_coords(proto_coords, len(truth_coords))\n            score = tm_like_score(pred_coords, truth_coords)\n            best_local = max(best_local, score)\n\n        target_scores.append(best_local)\n\n    mean_score = float(np.mean(target_scores)) if len(target_scores) > 0 else -np.inf\n\n    tuning_rows.append({\n        \"weights\": weights,\n        \"mean_val_tm_like\": mean_score\n    })\n\ntuning_df = pd.DataFrame(tuning_rows).sort_values(\"mean_val_tm_like\", ascending=False).reset_index(drop=True)\ndisplay(tuning_df)\nbest_weights = tuning_df.loc[0, \"weights\"]\nprint(\"Best weights:\", best_weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:14:00.88897Z","iopub.execute_input":"2026-03-24T20:14:00.890389Z","iopub.status.idle":"2026-03-24T20:36:56.541483Z","shell.execute_reply.started":"2026-03-24T20:14:00.890328Z","shell.execute_reply":"2026-03-24T20:36:56.540065Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Inspect top matches for the test set under the tuned weights","metadata":{}},{"cell_type":"code","source":"test_seq = test_seq.copy()\ntest_seq[\"sequence\"] = test_seq[\"sequence\"].astype(str).apply(normalize_sequence)\n\ntest_match_rows = []\n\nfor _, row in test_seq.iterrows():\n    target_id = row[\"target_id\"]\n    seq = row[\"sequence\"]\n\n    ranked = rank_training_templates(seq, best_weights, top_k=5)\n\n    out = {\n        \"test_target_id\": target_id,\n        \"test_length\": len(seq),\n    }\n\n    for i, (tid, score) in enumerate(ranked, start=1):\n        out[f\"match_{i}\"] = tid\n        out[f\"score_{i}\"] = score\n\n    test_match_rows.append(out)\n\ntest_match_df = pd.DataFrame(test_match_rows)\ndisplay(test_match_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T20:39:38.806847Z","iopub.execute_input":"2026-03-24T20:39:38.807221Z","iopub.status.idle":"2026-03-24T20:43:28.092333Z","shell.execute_reply.started":"2026-03-24T20:39:38.807194Z","shell.execute_reply":"2026-03-24T20:43:28.091138Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"build the submission using the top 5 matched templates","metadata":{}},{"cell_type":"code","source":"submission = sample_sub.copy()\nsubmission[\"target_id\"] = submission[\"ID\"].astype(str).apply(extract_target_id)\n\nmatch_lookup = {}\n\nfor _, row in test_match_df.iterrows():\n    target_id = row[\"test_target_id\"]\n    matched_list = []\n    for i in range(1, 6):\n        matched_tid = row[f\"match_{i}\"]\n        matched_list.append(matched_tid)\n    match_lookup[target_id] = matched_list\n\nparts = []\n\nfor target_id, g in submission.groupby(\"target_id\", sort=False):\n    g = g.copy().sort_values(\"resid\").reset_index(drop=True)\n    target_len = len(g)\n\n    matched_templates = match_lookup[target_id]\n\n    for pred_num in range(1, 6):\n        template_tid = matched_templates[pred_num - 1]\n        proto_coords = prototype_dict[template_tid][\"coords\"]\n        pred_coords = interpolate_coords(proto_coords, target_len)\n\n        g[f\"x_{pred_num}\"] = np.round(pred_coords[:, 0], 3)\n        g[f\"y_{pred_num}\"] = np.round(pred_coords[:, 1], 3)\n        g[f\"z_{pred_num}\"] = np.round(pred_coords[:, 2], 3)\n\n    parts.append(g)\n\nsubmission = pd.concat(parts, ignore_index=True)\nsubmission = submission.drop(columns=[\"target_id\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T21:00:21.333497Z","iopub.execute_input":"2026-03-24T21:00:21.334087Z","iopub.status.idle":"2026-03-24T21:00:21.440139Z","shell.execute_reply.started":"2026-03-24T21:00:21.334053Z","shell.execute_reply":"2026-03-24T21:00:21.439194Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"validate the submission format","metadata":{}},{"cell_type":"code","source":"required_cols = [\n    \"ID\", \"resname\", \"resid\",\n    \"x_1\", \"y_1\", \"z_1\",\n    \"x_2\", \"y_2\", \"z_2\",\n    \"x_3\", \"y_3\", \"z_3\",\n    \"x_4\", \"y_4\", \"z_4\",\n    \"x_5\", \"y_5\", \"z_5\"\n]\n\nprint(\"submission shape:\", submission.shape)\ndisplay(submission.head())\n\nassert list(submission.columns) == required_cols, \"Column order is incorrect.\"\nassert submission.isnull().sum().sum() == 0, \"Submission contains missing values.\"\n\nprint(\"Submission format looks valid.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T21:00:28.157544Z","iopub.execute_input":"2026-03-24T21:00:28.159326Z","iopub.status.idle":"2026-03-24T21:00:28.190365Z","shell.execute_reply.started":"2026-03-24T21:00:28.159268Z","shell.execute_reply":"2026-03-24T21:00:28.188991Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"save the Kaggle file","metadata":{}},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)\nprint(\"Saved submission.csv\")\n\n!head submission.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-24T21:00:31.942519Z","iopub.execute_input":"2026-03-24T21:00:31.944Z","iopub.status.idle":"2026-03-24T21:00:32.383891Z","shell.execute_reply.started":"2026-03-24T21:00:31.943873Z","shell.execute_reply":"2026-03-24T21:00:32.38271Z"}},"outputs":[],"execution_count":null}]}