{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":118765,"databundleVersionId":15231210,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install biopython -q\n\nimport os\nimport numpy as np\nimport pandas as pd\nfrom Bio import AlignIO\nfrom tqdm import tqdm\nimport shutil\n\n# --- 1. SET PATHS ---\n# Based on competition data, we need the MSA folder and train_sequences.csv \nINPUT_BASE = \"/kaggle/input/stanford-rna-3d-folding-2\"\nMSA_DIR = os.path.join(INPUT_BASE, \"MSA\")\nSEQ_FILE = os.path.join(INPUT_BASE, \"train_sequences.csv\")\nOUTPUT_DIR = \"/kaggle/working/evolutionary_features_npy\"\n\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\n# --- 2. DEBUGGING CHECK ---\nif not os.path.exists(MSA_DIR):\n    print(f\"❌ ERROR: MSA Directory not found at {MSA_DIR}\")\n    # List directory to see what's actually there\n    print(\"Available in input:\", os.listdir(INPUT_BASE))\nelse:\n    print(f\"✅ Found MSA Directory: {MSA_DIR}\")\n\n# --- 3. PROCESSING ---\ndef calculate_pssm(msa_path):\n    try:\n        alignment = AlignIO.read(msa_path, \"fasta\")\n        seq_len = alignment.get_alignment_length()\n        counts = np.zeros((seq_len, 4))\n        for i in range(seq_len):\n            col = alignment[:, i].upper()\n            counts[i, 0] = col.count('A')\n            counts[i, 1] = col.count('C')\n            counts[i, 2] = col.count('G')\n            counts[i, 3] = col.count('U') + col.count('T')\n        \n        sums = counts.sum(axis=1, keepdims=True)\n        sums[sums == 0] = 1.0 \n        pssm = counts / sums\n        return pssm.astype(np.float32)\n    except Exception as e:\n        return None\n\n# --- 4. EXECUTION ---\nif os.path.exists(SEQ_FILE):\n    df = pd.read_csv(SEQ_FILE).head(100) # Increased to 100 for a better test\n    processed_count = 0\n    \n    for target_id in tqdm(df['target_id'], desc=\"Processing MSAs\"):\n        msa_path = os.path.join(MSA_DIR, f\"{target_id}.MSA.fasta\")\n        if os.path.exists(msa_path):\n            features = calculate_pssm(msa_path)\n            if features is not None:\n                np.save(os.path.join(OUTPUT_DIR, f\"{target_id}.npy\"), features)\n                processed_count += 1\n    \n    # --- 5. THE \"FORCE REFRESH\" STEP ---\n    print(f\"\\n✅ SUCCESS: Processed {processed_count} files.\")\n    \n    if processed_count > 0:\n        print(\"Creating a ZIP archive to ensure output visibility...\")\n        shutil.make_archive(\"rna_evolutionary_data\", 'zip', OUTPUT_DIR)\n        print(\"Archive created: rna_evolutionary_data.zip\")\n    else:\n        print(\"⚠️ No files were processed. Check if target_id matches MSA filenames.\")\nelse:\n    print(f\"❌ CRITICAL ERROR: Could not find {SEQ_FILE}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null}]}