{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":118765,"databundleVersionId":15231210}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nimport numpy as np\nimport pandas as pd\nimport os, re, math\nfrom collections import defaultdict\n\n\ninput_dir = \"/kaggle/input\"\ntest_csv = None\nfor root, dirs, files in os.walk(input_dir):\n    for f in files:\n        if f == \"test_sequences.csv\":\n            test_csv = os.path.join(root, f)\n            break\n\nif test_csv is None:\n    raise FileNotFoundError(\"test_sequences.csv not found in /kaggle/input/\")\n\nprint(f\"Found: {test_csv}\")\ndf = pd.read_csv(test_csv)\nprint(df.head())\nprint(f\"Columns: {df.columns.tolist()}\")\nprint(f\"Rows: {len(df)}\")\n\n# ── 2. DETECT COLUMN NAMES ──────────────────────────────────\n# Handle possible column name variants\nseq_col  = next((c for c in df.columns if 'seq' in c.lower()), None)\nid_col   = next((c for c in df.columns if 'id'  in c.lower()), None)\n\nif seq_col is None or id_col is None:\n    print(\"Available columns:\", df.columns.tolist())\n    raise ValueError(\"Cannot find sequence or ID column. Check column names above.\")\n\nprint(f\"Using ID col: '{id_col}', Sequence col: '{seq_col}'\")\n\n\nPAIRS = {('A','U'),('U','A'),('G','C'),('C','G'),('G','U'),('U','G')}\nMIN_LOOP = 3\n\ndef nussinov(seq):\n    n = len(seq)\n    dp = [[0]*n for _ in range(n)]\n    for gap in range(MIN_LOOP+1, n):\n        for i in range(n - gap):\n            j = i + gap\n            dp[i][j] = dp[i][j-1]\n            for k in range(i, j - MIN_LOOP):\n                if (seq[k], seq[j]) in PAIRS:\n                    val = (dp[i][k-1] if k>i else 0) + dp[k+1][j-1] + 1\n                    if val > dp[i][j]:\n                        dp[i][j] = val\n    # traceback\n    pairs = [-1]*n\n    stack = [(0, n-1)]\n    while stack:\n        i, j = stack.pop()\n        if i >= j:\n            continue\n        if dp[i][j] == dp[i][j-1]:\n            stack.append((i, j-1))\n        else:\n            for k in range(i, j - MIN_LOOP):\n                if (seq[k], seq[j]) in PAIRS:\n                    if (dp[i][k-1] if k>i else 0) + dp[k+1][j-1] + 1 == dp[i][j]:\n                        pairs[k] = j\n                        pairs[j] = k\n                        if k > i: stack.append((i, k-1))\n                        stack.append((k+1, j-1))\n                        break\n    return pairs\n\n# ── 4. 3D COORDINATE BUILDER ─────────────────────────────────\n# A-form RNA helix parameters\nRISE     = 2.81    # Å per residue along helix axis\nTWIST    = 32.7    # degrees per residue\nRADIUS   = 9.0     # Å (C1' from helix axis in A-form)\n# Single-strand extended chain\nSS_RISE  = 5.9     # Å between consecutive C1' atoms in ssRNA\n\ndef build_coords(seq, pairs, noise_scale=0.0, rng=None):\n    \"\"\"\n    Build C1' atom coordinates.\n    Paired residues → A-form double helix geometry.\n    Unpaired residues → extended single-strand chain.\n    Returns np.array shape (N, 3).\n    \"\"\"\n    if rng is None:\n        rng = np.random.default_rng(42)\n\n    n = len(seq)\n    coords = np.zeros((n, 3))\n\n    # Assign helix IDs: consecutive paired runs get same helix\n    helix_id = [-1] * n\n    hid = 0\n    visited = set()\n    for i in range(n):\n        j = pairs[i]\n        if j > i and i not in visited:\n            # trace helix run\n            a, b = i, j\n            while a < b and pairs[a] == b and a not in visited:\n                helix_id[a] = hid\n                helix_id[b] = hid\n                visited.add(a)\n                visited.add(b)\n                a += 1\n                b -= 1\n            hid += 1\n\n    # Build helix segments + place ss regions\n    placed = [False] * n\n    # Simple sequential placement: walk from 5' to 3'\n    # Track current \"cursor\" position + direction\n    pos = np.array([0.0, 0.0, 0.0])\n    direction = np.array([0.0, 0.0, 1.0])\n\n    # For helix segments, we pre-compute axis, then place both strands\n    helix_starts = {}  # hid -> first residue index (5' side)\n    for i in range(n):\n        h = helix_id[i]\n        if h >= 0:\n            if h not in helix_starts or i < helix_starts[h]:\n                helix_starts[h] = i\n\n    # Walk residues in order\n    i = 0\n    cursor = np.array([0.0, 0.0, 0.0])\n    while i < n:\n        h = helix_id[i]\n        if h >= 0 and not placed[i]:\n            # Find helix extent on 5' strand\n            a = i\n            helix_res_5 = []\n            helix_res_3 = []\n            while a < n and helix_id[a] == h and not placed[a]:\n                b = pairs[a]\n                helix_res_5.append(a)\n                helix_res_3.append(b)\n                placed[a] = True\n                placed[b] = True\n                a += 1\n            length = len(helix_res_5)\n            # Helix axis starts at cursor, goes along z\n            axis = cursor.copy()\n            twist_rad = math.radians(TWIST)\n            for step, (r5, r3) in enumerate(zip(helix_res_5, helix_res_3)):\n                angle5 = step * twist_rad\n                angle3 = angle5 + math.pi  # opposite strand\n                z = axis[2] + step * RISE\n                coords[r5] = [\n                    axis[0] + RADIUS * math.cos(angle5),\n                    axis[1] + RADIUS * math.sin(angle5),\n                    z\n                ]\n                coords[r3] = [\n                    axis[0] + RADIUS * math.cos(angle3),\n                    axis[1] + RADIUS * math.sin(angle3),\n                    z\n                ]\n            # Advance cursor to end of helix\n            cursor = np.array([axis[0], axis[1], axis[2] + length * RISE])\n            i = a\n        else:\n            # Single-stranded residue\n            if not placed[i]:\n                coords[i] = cursor.copy()\n                placed[i] = True\n                cursor = cursor + np.array([0.0, SS_RISE * 0.3, SS_RISE * 0.95])\n            i += 1\n\n    # Add Gaussian noise for diversity\n    if noise_scale > 0:\n        coords += rng.normal(0, noise_scale, coords.shape)\n\n    return coords\n\n\n# ── 5. BUILD NUCLEOTIDE NAME MAP ─────────────────────────────\nRESNAME_MAP = {'A':'A','U':'U','G':'G','C':'C',\n               'a':'A','u':'U','g':'G','c':'C',\n               'T':'U','t':'U'}  # treat T as U for RNA\n\n# ── 6. GENERATE PREDICTIONS ──────────────────────────────────\nNOISE_LEVELS = [0.0, 0.4, 0.8, 1.2, 1.6]   # 5 prediction perturbations\n\nrows = []\nfor _, row in df.iterrows():\n    seq_id  = str(row[id_col])\n    seq_raw = str(row[seq_col]).strip().upper().replace('T','U')\n    seq     = seq_raw\n\n    # Compute secondary structure\n    pairs = nussinov(seq)\n\n    # Generate 5 coordinate sets\n    all_coords = []\n    for k, noise in enumerate(NOISE_LEVELS):\n        rng = np.random.default_rng(hash(seq_id + str(k)) % (2**31))\n        xyz = build_coords(seq, pairs, noise_scale=noise, rng=rng)\n        all_coords.append(xyz)\n\n    # Write one row per residue\n    for res_idx, nt in enumerate(seq):\n        resname = RESNAME_MAP.get(nt, 'N')\n        resid   = res_idx + 1\n        row_data = {\n            \"ID\":      f\"{seq_id}_{resid}\",   # e.g. R1107_1\n            \"resname\": resname,\n            \"resid\":   resid,\n        }\n        for k in range(5):\n            x, y, z = all_coords[k][res_idx]\n            row_data[f\"x_{k+1}\"] = round(float(x), 3)\n            row_data[f\"y_{k+1}\"] = round(float(y), 3)\n            row_data[f\"z_{k+1}\"] = round(float(z), 3)\n        rows.append(row_data)\n\n# ── 7. WRITE submission.csv ──────────────────────────────────\ncols = [\"ID\",\"resname\",\"resid\"]\nfor k in range(1,6):\n    cols += [f\"x_{k}\", f\"y_{k}\", f\"z_{k}\"]\n\nsubmission = pd.DataFrame(rows, columns=cols)\nout_path = \"/kaggle/working/submission.csv\"\nsubmission.to_csv(out_path, index=False)\n\nprint(f\"\\n✅ submission.csv saved → {out_path}\")\nprint(f\"   Shape: {submission.shape}\")\nprint(submission.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-28T10:46:05.262496Z","iopub.execute_input":"2026-02-28T10:46:05.262774Z"}},"outputs":[],"execution_count":null}]}