{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":14604295,"datasetId":9328538,"databundleVersionId":15440074},{"sourceType":"kernelVersion","sourceId":227713340,"isSourceIdPinned":false}],"dockerImageVersionId":31329,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"##### References:\n###### - Template-Based Modeling approach inspired by public notebooks in Stanford RNA 3D Folding Part 2\n###### - RhoFold+: Shen et al., Nature Methods 2024 (https://doi.org/10.1038/s41592-024-02487-0)\n###### - RhoFold pretrained weights: https://www.kaggle.com/code/ogurtsov/rhofold-ribonanzanet-msas-lb-0-215","metadata":{}},{"cell_type":"markdown","source":"## Cell 1 Setup","metadata":{}},{"cell_type":"code","source":"# References:\n# - Template-Based Modeling approach inspired by public notebooks in Stanford RNA 3D Folding Part 2\n# - RhoFold+: Shen et al., Nature Methods 2024 (https://doi.org/10.1038/s41592-024-02487-0)\n# - RhoFold pretrained weights: https://www.kaggle.com/code/ogurtsov/rhofold-ribonanzanet-msas-lb-0-215\n\nimport os\nimport sys\nimport subprocess\n\n# Install biopython offline\nwhl_file = '/kaggle/input/datasets/kami1976/biopython-cp312/biopython-1.86-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl'\nsubprocess.run([sys.executable, '-m', 'pip', 'install', whl_file, '-q'], check=True)\nprint(\"biopython installed!\")\n\n# Add RhoFold to path\nRHOFOLD_PATH = '/kaggle/input/notebooks/ogurtsov/rhofold-ribonanzanet-msas-lb-0-215/RhoFold'\nsys.path.insert(0, RHOFOLD_PATH)\n\ntry:\n    from rhofold.rhofold import RhoFold\n    from rhofold.config import rhofold_config\n    print(\"RhoFold imported successfully!\")\nexcept Exception as e:\n    print(\"Import error:\", e)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-21T20:07:03.980964Z","iopub.execute_input":"2026-03-21T20:07:03.981352Z","iopub.status.idle":"2026-03-21T20:07:07.084538Z","shell.execute_reply.started":"2026-03-21T20:07:03.981315Z","shell.execute_reply":"2026-03-21T20:07:07.083864Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Cell 2 Imports & Constants","metadata":{}},{"cell_type":"code","source":"import gc\nimport os\nimport sys\nimport time\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport torch\nfrom Bio.Align import PairwiseAligner\nfrom tqdm import tqdm\n\n# ── Kaggle mode detection ─────────────────────────────────────────────────────\nIS_KAGGLE = bool(os.environ.get(\"KAGGLE_IS_COMPETITION_RERUN\", \"\"))\nLOCAL_N_SAMPLES = 2\nif IS_KAGGLE:\n    print(\"Running in KAGGLE COMPETITION mode.\")\nelse:\n    print(f\"Running in LOCAL mode — only first {LOCAL_N_SAMPLES} targets.\")\n\n# ── Paths ─────────────────────────────────────────────────────────────────────\nDATA_BASE = \"/kaggle/input/competitions/stanford-rna-3d-folding-2\"\nTEST_CSV   = f\"{DATA_BASE}/test_sequences.csv\"\nTRAIN_CSV  = f\"{DATA_BASE}/train_sequences.csv\"\nTRAIN_LBLS = f\"{DATA_BASE}/train_labels.csv\"\nVAL_CSV    = f\"{DATA_BASE}/validation_sequences.csv\"\nVAL_LBLS   = f\"{DATA_BASE}/validation_labels.csv\"\nOUTPUT_CSV = \"/kaggle/working/submission.csv\"\nMSA_DIR    = f\"{DATA_BASE}/MSA\"\nTEMP_DIR   = \"/kaggle/working/temp_fasta\"\nos.makedirs(TEMP_DIR, exist_ok=True)\n\nRHOFOLD_CKPT = '/kaggle/input/notebooks/ogurtsov/rhofold-ribonanzanet-msas-lb-0-215/RhoFold/pretrained/RhoFold_pretrained.pt'\n\n# ── Hyperparameters ───────────────────────────────────────────────────────────\nN_SAMPLE             = 5\nSEED                 = 42\nMIN_PERCENT_IDENTITY = 40.0  # lowered from 50 to find more templates\nRHOFOLD_MAX_LEN      = 400   # skip RhoFold for sequences longer than this","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T20:07:07.085691Z","iopub.execute_input":"2026-03-21T20:07:07.085923Z","iopub.status.idle":"2026-03-21T20:07:07.09306Z","shell.execute_reply.started":"2026-03-21T20:07:07.0859Z","shell.execute_reply":"2026-03-21T20:07:07.092149Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Cell 3 Helper Functions","metadata":{}},{"cell_type":"code","source":"def coords_to_rows(target_id, seq, coords):\n    \"\"\"coords shape: (N_SAMPLE, seq_len, 3)\"\"\"\n    rows = []\n    for i in range(len(seq)):\n        row = {\"ID\": f\"{target_id}_{i+1}\", \"resname\": seq[i], \"resid\": i+1}\n        for s in range(N_SAMPLE):\n            if s < coords.shape[0] and i < coords.shape[1]:\n                x, y, z = coords[s, i]\n            else:\n                x, y, z = 0.0, 0.0, 0.0\n            row[f\"x_{s+1}\"] = float(x)\n            row[f\"y_{s+1}\"] = float(y)\n            row[f\"z_{s+1}\"] = float(z)\n        rows.append(row)\n    return rows\n\ndef generate_helix_fallback(seq, seed=None):\n    \"\"\"A-form RNA helix fallback when no template or RhoFold available.\"\"\"\n    if seed is not None:\n        np.random.seed(seed)\n    n = len(seq)\n    coords = np.zeros((n, 3))\n    for i in range(n):\n        ang = i * 0.6\n        coords[i] = [10.0 * np.cos(ang), 10.0 * np.sin(ang), i * 2.5]\n    return coords\n\ndef process_labels(labels_df):\n    \"\"\"Build target_id -> coords lookup from labels dataframe.\"\"\"\n    coords = {}\n    prefixes = labels_df[\"ID\"].str.rsplit(\"_\", n=1).str[0]\n    for prefix, grp in labels_df.groupby(prefixes):\n        coords[prefix] = grp.sort_values(\"resid\")[[\"x_1\",\"y_1\",\"z_1\"]].values\n    return coords\n\ndef parse_fasta(fasta_content):\n    out, cur, parts = {}, None, []\n    for line in str(fasta_content).splitlines():\n        line = line.strip()\n        if not line:\n            continue\n        if line.startswith(\">\"):\n            if cur is not None:\n                out[cur] = \"\".join(parts)\n            cur = line[1:].split()[0]\n            parts = []\n        else:\n            parts.append(line.replace(\" \", \"\"))\n    if cur is not None:\n        out[cur] = \"\".join(parts)\n    return out\n\ndef parse_stoichiometry(stoich):\n    if pd.isna(stoich) or str(stoich).strip() == \"\":\n        return []\n    return [(ch.strip(), int(cnt)) for part in str(stoich).split(\";\")\n            for ch, cnt in [part.split(\":\")]]\n\ndef get_chain_segments(row):\n    seq = row[\"sequence\"]\n    stoich = row.get(\"stoichiometry\", \"\")\n    all_sq = row.get(\"all_sequences\", \"\")\n    if pd.isna(stoich) or pd.isna(all_sq) or str(stoich).strip() == \"\":\n        return [(0, len(seq))]\n    try:\n        chain_dict = parse_fasta(all_sq)\n        order = parse_stoichiometry(stoich)\n        segs, pos = [], 0\n        for ch, cnt in order:\n            base = chain_dict.get(ch)\n            if base is None:\n                return [(0, len(seq))]\n            for _ in range(cnt):\n                segs.append((pos, pos + len(base)))\n                pos += len(base)\n        return segs if pos == len(seq) else [(0, len(seq))]\n    except Exception:\n        return [(0, len(seq))]\n\ndef build_segments_map(df):\n    return {r[\"target_id\"]: get_chain_segments(r) for _, r in df.iterrows()}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T20:07:07.094028Z","iopub.execute_input":"2026-03-21T20:07:07.094334Z","iopub.status.idle":"2026-03-21T20:07:07.112293Z","shell.execute_reply.started":"2026-03-21T20:07:07.094311Z","shell.execute_reply":"2026-03-21T20:07:07.111642Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Cell 4 TBM Functions","metadata":{}},{"cell_type":"code","source":"def _make_aligner():\n    al = PairwiseAligner()\n    al.mode             = \"global\"\n    al.match_score      = 2\n    al.mismatch_score   = -1.5\n    al.open_gap_score   = -8\n    al.extend_gap_score = -0.4\n    return al\n\n_aligner = _make_aligner()\n\ndef find_similar_sequences(query_seq, train_seqs_df, train_coords_dict, top_n=30):\n    \"\"\"Find top-n most similar training sequences using pairwise alignment.\"\"\"\n    results = []\n    for _, row in train_seqs_df.iterrows():\n        tid, tseq = row[\"target_id\"], row[\"sequence\"]\n        if tid not in train_coords_dict:\n            continue\n        if abs(len(tseq) - len(query_seq)) / max(len(tseq), len(query_seq)) > 0.3:\n            continue\n        aln    = next(iter(_aligner.align(query_seq, tseq)))\n        norm_s = aln.score / (2 * min(len(query_seq), len(tseq)))\n        identical = sum(\n            1 for (qs, qe), (ts, te) in zip(*aln.aligned)\n            for qp, tp in zip(range(qs, qe), range(ts, te))\n            if query_seq[qp] == tseq[tp]\n        )\n        pct_id = 100 * identical / len(query_seq)\n        results.append((tid, tseq, norm_s, train_coords_dict[tid], pct_id))\n    results.sort(key=lambda x: x[2], reverse=True)\n    return results[:top_n]\n\ndef adapt_template_to_query(query_seq, template_seq, template_coords):\n    \"\"\"Map template coordinates onto query sequence using alignment.\"\"\"\n    aln = next(iter(_aligner.align(query_seq, template_seq)))\n    new_coords = np.full((len(query_seq), 3), np.nan)\n    for (qs, qe), (ts, te) in zip(*aln.aligned):\n        chunk = template_coords[ts:te]\n        if len(chunk) == (qe - qs):\n            new_coords[qs:qe] = chunk\n    for i in range(len(new_coords)):\n        if np.isnan(new_coords[i, 0]):\n            pv = next((j for j in range(i-1, -1, -1) if not np.isnan(new_coords[j, 0])), -1)\n            nv = next((j for j in range(i+1, len(new_coords)) if not np.isnan(new_coords[j, 0])), -1)\n            if pv >= 0 and nv >= 0:\n                w = (i - pv) / (nv - pv)\n                new_coords[i] = (1 - w) * new_coords[pv] + w * new_coords[nv]\n            elif pv >= 0:\n                new_coords[i] = new_coords[pv] + [3, 0, 0]\n            elif nv >= 0:\n                new_coords[i] = new_coords[nv] + [3, 0, 0]\n            else:\n                new_coords[i] = [i * 3, 0, 0]\n    return np.nan_to_num(new_coords)\n\ndef apply_diversity_transform(coords, slot, segs, rng, sim):\n    \"\"\"Apply diversity transforms to generate structurally varied predictions.\"\"\"\n    if slot == 0:\n        return coords\n    elif slot == 1:\n        return coords + rng.normal(0, max(0.01, (0.40 - sim) * 0.06), coords.shape)\n    elif slot == 2:\n        longest = max(segs, key=lambda se: se[1] - se[0])\n        s, e = longest\n        L = e - s\n        if L < 30:\n            return coords\n        pivot = s + int(rng.integers(10, L - 10))\n        a = rng.normal(size=3); a /= np.linalg.norm(a) + 1e-12\n        ang = np.deg2rad(float(rng.uniform(-22, 22)))\n        x, y, z = a; c, sc = np.cos(ang), np.sin(ang); CC = 1 - c\n        R = np.array([[c+x*x*CC, x*y*CC-z*sc, x*z*CC+y*sc],\n                      [y*x*CC+z*sc, c+y*y*CC, y*z*CC-x*sc],\n                      [z*x*CC-y*sc, z*y*CC+x*sc, c+z*z*CC]])\n        X = coords.copy(); p0 = X[pivot].copy()\n        X[pivot+1:e] = (X[pivot+1:e] - p0) @ R.T + p0\n        return X\n    elif slot == 3:\n        X = coords.copy(); gc_ = X.mean(0, keepdims=True)\n        for s, e in segs:\n            a = rng.normal(size=3); a /= np.linalg.norm(a) + 1e-12\n            ang = np.deg2rad(float(rng.uniform(-12, 12)))\n            x, y, z = a; c, sc = np.cos(ang), np.sin(ang); CC = 1 - c\n            R = np.array([[c+x*x*CC, x*y*CC-z*sc, x*z*CC+y*sc],\n                          [y*x*CC+z*sc, c+y*y*CC, y*z*CC-x*sc],\n                          [z*x*CC-y*sc, z*y*CC+x*sc, c+z*z*CC]])\n            shift = rng.normal(size=3)\n            shift = shift / (np.linalg.norm(shift) + 1e-12) * float(rng.uniform(0, 1.5))\n            c_ = X[s:e].mean(0, keepdims=True)\n            X[s:e] = (X[s:e] - c_) @ R.T + c_ + shift\n        X -= X.mean(0, keepdims=True) - gc_\n        return X\n    else:\n        X = coords.copy()\n        for s, e in segs:\n            L = e - s\n            if L < 20:\n                continue\n            ctrl = np.linspace(0, L-1, 6)\n            disp = rng.normal(0, 0.8, (6, 3))\n            t = np.arange(L)\n            X[s:e] += np.vstack([np.interp(t, ctrl, disp[:, k]) for k in range(3)]).T\n        return X","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T20:07:07.113842Z","iopub.execute_input":"2026-03-21T20:07:07.114085Z","iopub.status.idle":"2026-03-21T20:07:07.134296Z","shell.execute_reply.started":"2026-03-21T20:07:07.114064Z","shell.execute_reply":"2026-03-21T20:07:07.133731Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Cell 5 RhoFold Functions","metadata":{}},{"cell_type":"code","source":"VALID_RNA_CHARS = set('AUGCTaugct-Nn')\n\ndef clean_msa_fasta(input_path, output_path):\n    \"\"\"Clean MSA fasta to only contain valid RNA characters.\"\"\"\n    cleaned_seqs = []\n    current_header, current_seq = None, []\n    with open(input_path, 'r') as f:\n        for line in f:\n            line = line.rstrip()\n            if line.startswith('>'):\n                if current_header and current_seq:\n                    cleaned_seqs.append((current_header, ''.join(current_seq)))\n                current_header, current_seq = line, []\n            else:\n                cleaned = ''.join(c if c in VALID_RNA_CHARS else 'N' for c in line)\n                current_seq.append(cleaned)\n    if current_header and current_seq:\n        cleaned_seqs.append((current_header, ''.join(current_seq)))\n    with open(output_path, 'w') as f:\n        for header, seq in cleaned_seqs:\n            f.write(f'{header}\\n{seq}\\n')\n\ndef load_rhofold_model(ckpt_path, device):\n    \"\"\"Load pretrained RhoFold model.\"\"\"\n    from rhofold.rhofold import RhoFold as RhoFoldModel\n    from rhofold.config import rhofold_config\n    model = RhoFoldModel(rhofold_config)\n    model.load_state_dict(\n        torch.load(ckpt_path, map_location=device)['model'], strict=False)\n    model = model.to(device)\n    model.eval()\n    return model\n\ndef rhofold_predict(model, tid, seq, device):\n    \"\"\"Run RhoFold inference for one RNA sequence using MSA.\"\"\"\n    from rhofold.utils.alphabet import get_features\n    try:\n        fas_path = os.path.join(TEMP_DIR, f'{tid}.fasta')\n        with open(fas_path, 'w') as f:\n            f.write(f'>{tid}\\n{seq}\\n')\n        msa_path       = os.path.join(MSA_DIR, f'{tid}.MSA.fasta')\n        clean_msa_path = os.path.join(TEMP_DIR, f'{tid}_clean.fasta')\n        clean_msa_fasta(msa_path, clean_msa_path)\n        data          = get_features(fas_path, clean_msa_path, msa_depth=128)\n        tokens        = data['tokens'].to(device)\n        rna_fm_tokens = data['rna_fm_tokens'].to(device)\n        rf_seq        = data['seq']\n        with torch.no_grad():\n            outputs = model(tokens, rna_fm_tokens, rf_seq)\n        c1_coords = outputs[-1][\"cords_c1'\"][-1].squeeze(0).cpu().numpy()\n        torch.cuda.empty_cache()\n        return c1_coords\n    except Exception as e:\n        print(f\"  RhoFold failed for {tid}: {e}\")\n        torch.cuda.empty_cache()\n        return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T20:07:07.13504Z","iopub.execute_input":"2026-03-21T20:07:07.135303Z","iopub.status.idle":"2026-03-21T20:07:07.148844Z","shell.execute_reply.started":"2026-03-21T20:07:07.135282Z","shell.execute_reply":"2026-03-21T20:07:07.148253Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Cell 6 Main","metadata":{}},{"cell_type":"code","source":"def main():\n    import torch\n\n    np.random.seed(SEED)\n    torch.manual_seed(SEED)\n\n    # ── Load data ─────────────────────────────────────────────────────────────\n    test_df_full = pd.read_csv(TEST_CSV)\n    test_df = (test_df_full.head(LOCAL_N_SAMPLES) if not IS_KAGGLE\n               else test_df_full).reset_index(drop=True)\n    print(f\"Test targets: {len(test_df)}\")\n\n    print(\"\\nLoading training data...\")\n    train_seqs   = pd.read_csv(TRAIN_CSV)\n    val_seqs     = pd.read_csv(VAL_CSV)\n    train_labels = pd.read_csv(TRAIN_LBLS, low_memory=False)\n    val_labels   = pd.read_csv(VAL_LBLS,   low_memory=False)\n\n    combined_seqs   = pd.concat([train_seqs, val_seqs],     ignore_index=True)\n    combined_labels = pd.concat([train_labels, val_labels], ignore_index=True)\n    train_coords    = process_labels(combined_labels)\n    segments_map    = build_segments_map(test_df)\n    print(f\"Template pool: {len(combined_seqs)} sequences, {len(train_coords)} structures\")\n\n    # ── PHASE 1: TBM ──────────────────────────────────────────────────────────\n    print(f\"\\n{'='*60}\")\n    print(f\"PHASE 1: Template-Based Modeling (MIN_PCT_IDENTITY={MIN_PERCENT_IDENTITY})\")\n    print(f\"{'='*60}\")\n    t0 = time.time()\n\n    tbm_preds     = {}\n    rhofold_queue = []\n\n    for _, row in test_df.iterrows():\n        tid  = row[\"target_id\"]\n        seq  = row[\"sequence\"]\n        segs = segments_map.get(tid, [(0, len(seq))])\n\n        similar = find_similar_sequences(seq, combined_seqs, train_coords, top_n=30)\n        preds, used = [], set()\n\n        for i, (tmpl_id, tmpl_seq, sim, tmpl_coords, pct_id) in enumerate(similar):\n            if len(preds) >= N_SAMPLE:\n                break\n            if pct_id < MIN_PERCENT_IDENTITY:\n                break\n            if tmpl_id in used:\n                continue\n            rng     = np.random.default_rng((row.name * 10000000000 + i * 10007) % (2**32))\n            adapted = adapt_template_to_query(seq, tmpl_seq, tmpl_coords)\n            X       = apply_diversity_transform(adapted, len(preds), segs, rng, sim)\n            preds.append(X)\n            used.add(tmpl_id)\n\n        tbm_preds[tid] = preds\n        if len(seq) <= RHOFOLD_MAX_LEN:\n            rhofold_queue.append((tid, seq))\n        print(f\"  {tid} ({len(seq)} nt): {len(preds)} TBM predictions\")\n\n    print(f\"\\nPhase 1 done in {time.time()-t0:.1f}s\")\n\n    # ── PHASE 2: RhoFold inference ────────────────────────────────────────────\n    print(f\"\\n{'='*60}\")\n    print(\"PHASE 2: RhoFold Deep Learning Inference\")\n    print(f\"{'='*60}\")\n\n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n    print(f\"Loading RhoFold model on {device}...\")\n    rf_model = load_rhofold_model(RHOFOLD_CKPT, device)\n    print(\"RhoFold model loaded!\")\n\n    rhofold_preds = {}\n    for tid, seq in rhofold_queue:\n        print(f\"  Predicting {tid} (len={len(seq)})...\")\n        coords = rhofold_predict(rf_model, tid, seq, device)\n        if coords is not None:\n            rhofold_preds[tid] = coords\n            print(f\"    -> Success! shape={coords.shape}\")\n\n    print(f\"\\nRhoFold predictions: {len(rhofold_preds)}/{len(rhofold_queue)} sequences\")\n\n    # ── PHASE 3: Combine TBM + RhoFold ───────────────────────────────────────\n    print(f\"\\n{'='*60}\")\n    print(\"PHASE 3: Combine TBM + RhoFold\")\n    print(f\"{'='*60}\")\n\n    all_rows = []\n\n    for _, row in test_df.iterrows():\n        tid = row[\"target_id\"]\n        seq = row[\"sequence\"]\n\n        combined = list(tbm_preds.get(tid, []))\n\n        # If TBM found fewer than N_SAMPLE templates, fill with RhoFold or helix\n        if tid in rhofold_preds and len(combined) < N_SAMPLE:\n            combined.append(rhofold_preds[tid])\n            print(f\"  {tid}: RhoFold added as slot {len(combined)}\")\n\n        # Fill any remaining slots with helix fallback\n        while len(combined) < N_SAMPLE:\n            seed_val = row.name * 1000000 + len(combined) * 1000\n            combined.append(generate_helix_fallback(seq, seed=seed_val))\n\n        # If TBM filled all 5 slots, replace last slot with RhoFold for diversity\n        if tid in rhofold_preds and len(combined) >= N_SAMPLE:\n            combined[N_SAMPLE - 1] = rhofold_preds[tid]\n            print(f\"  {tid}: slot {N_SAMPLE} replaced with RhoFold\")\n\n        stacked = np.stack(combined[:N_SAMPLE], axis=0)\n        all_rows.extend(coords_to_rows(tid, seq, stacked))\n\n    # ── Save ──────────────────────────────────────────────────────────────────\n    sub = pd.DataFrame(all_rows)\n    cols = [\"ID\", \"resname\", \"resid\"] + [\n        f\"{c}_{i}\" for i in range(1, N_SAMPLE+1) for c in [\"x\", \"y\", \"z\"]\n    ]\n    coord_cols = [c for c in cols if c.startswith((\"x_\", \"y_\", \"z_\"))]\n    sub[coord_cols] = sub[coord_cols].clip(-999.999, 9999.999)\n    sub[cols].to_csv(OUTPUT_CSV, index=False)\n    print(f\"\\n✓ Saved {len(sub):,} rows to {OUTPUT_CSV}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T20:07:07.149679Z","iopub.execute_input":"2026-03-21T20:07:07.149906Z","iopub.status.idle":"2026-03-21T20:07:07.165436Z","shell.execute_reply.started":"2026-03-21T20:07:07.149871Z","shell.execute_reply":"2026-03-21T20:07:07.164726Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T20:07:07.166346Z","iopub.execute_input":"2026-03-21T20:07:07.166679Z","iopub.status.idle":"2026-03-21T20:07:48.826163Z","shell.execute_reply.started":"2026-03-21T20:07:07.166636Z","shell.execute_reply":"2026-03-21T20:07:48.825568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#read submission.csv\nsubmission_path = \"/kaggle/working/submission.csv\"\nsubmission_df = pd.read_csv(submission_path)\nprint(submission_df.head(20))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-21T20:07:48.826973Z","iopub.execute_input":"2026-03-21T20:07:48.827261Z","iopub.status.idle":"2026-03-21T20:07:48.845764Z","shell.execute_reply.started":"2026-03-21T20:07:48.827226Z","shell.execute_reply":"2026-03-21T20:07:48.845239Z"}},"outputs":[],"execution_count":null}]}