{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-07-18T07:10:12.274308Z","iopub.execute_input":"2026-07-18T07:10:12.274548Z","iopub.status.idle":"2026-07-18T07:10:14.616107Z","shell.execute_reply.started":"2026-07-18T07:10:12.274512Z","shell.execute_reply":"2026-07-18T07:10:14.613673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install biopython -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-18T07:10:14.618432Z","iopub.execute_input":"2026-07-18T07:10:14.618935Z","iopub.status.idle":"2026-07-18T07:10:22.155093Z","shell.execute_reply.started":"2026-07-18T07:10:14.618898Z","shell.execute_reply":"2026-07-18T07:10:22.154169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom itertools import product\nfrom Bio import SeqIO\nfrom scipy.sparse import csr_matrix\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.multioutput import MultiOutputClassifier\nimport joblib\n\n# =====================================================================\n# 1. SETUP THE 3-MER FEATURIZER\n# =====================================================================\nAMINO_ACIDS = \"ACDEFGHIKLMNPQRSTVWY\"\nTHREE_MERS = [\"\".join(p) for p in product(AMINO_ACIDS, repeat=3)]\nTHREE_MER_TO_IDX = {mer: i for i, mer in enumerate(THREE_MERS)}\n\ndef extract_3mer_features(fasta_path):\n    \"\"\"\n    Scans sequences to construct an N x 8000 normalized frequency matrix.\n    Uses sparse representations to drastically reduce RAM usage.\n    \"\"\"\n    print(f\"Parsing 3-mer frequencies from {fasta_path}...\")\n    indptr = [0]\n    indices = []\n    data = []\n    protein_ids = []\n    \n    for record in SeqIO.parse(fasta_path, \"fasta\"):\n        seq = str(record.seq).upper()\n        total_3mers = len(seq) - 2\n        protein_ids.append(record.id)\n        \n        if total_3mers <= 0:\n            indptr.append(len(indices))\n            continue\n            \n        local_counts = {}\n        for i in range(total_3mers):\n            mer = seq[i:i+3]\n            if mer in THREE_MER_TO_IDX:\n                mer_idx = THREE_MER_TO_IDX[mer]\n                local_counts[mer_idx] = local_counts.get(mer_idx, 0) + 1\n                \n        for mer_idx, count in local_counts.items():\n            indices.append(mer_idx)\n            data.append(count / total_3mers)\n        indptr.append(len(indices))\n        \n    X_sparse = csr_matrix((data, indices, indptr), shape=(len(protein_ids), 8000), dtype=np.float32)\n    return X_sparse, protein_ids\n\n# =====================================================================\n# 2. TARGET PREPARATION (CAFA 5 MULTI-LABEL)\n# =====================================================================\ndef load_labels(tsv_path, protein_ids, top_n_terms=1500):\n    \"\"\"\n    Builds the binary target matrix Y for the top N most frequent GO terms.\n    \"\"\"\n    print(\"Loading CAFA 5 training annotations...\")\n    train_terms = pd.read_csv(tsv_path, sep=\"\\t\")\n    \n    frequent_terms = train_terms['term'].value_counts().index[:top_n_terms].tolist()\n    term_to_idx = {term: i for i, term in enumerate(frequent_terms)}\n    \n    Y = np.zeros((len(protein_ids), top_n_terms), dtype=np.float32)\n    protein_to_idx = {pid: i for i, pid in enumerate(protein_ids)}\n    \n    for row in train_terms.itertuples():\n        if row.EntryID in protein_to_idx and row.term in term_to_idx:\n            p_idx = protein_to_idx[row.EntryID]\n            t_idx = term_to_idx[row.term]\n            Y[p_idx, t_idx] = 1.0\n            \n    return Y, frequent_terms\n\n# =====================================================================\n# 3. TRAINING AND SAVING PIPELINE\n# =====================================================================\nif __name__ == \"__main__\":\n    # Update paths to match your current environment\n    TRAIN_FASTA = \"/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_sequences.fasta\"\n    TRAIN_TERMS = \"/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_terms.tsv\"\n    \n    # Target filenames for saving your weights\n    MODEL_FILE = \"lr3mer_cafa5_model.pkl\"\n    METADATA_FILE = \"lr3mer_metadata.pkl\"\n    \n    # 1. Process features and targets\n    X_train, train_ids = extract_3mer_features(TRAIN_FASTA)\n    Y_train, selected_go_terms = load_labels(TRAIN_TERMS, train_ids, top_n_terms=1500)\n    \n    # 2. Setup model configurations\n    base_lr = LogisticRegression(max_iter=500, solver='saga', penalty='l2', C=1.0)\n    clf = MultiOutputClassifier(base_lr, n_jobs=-1)\n    \n    # 3. Fit weights\n    print(\"Training GOLabeler-style LR-3mer model...\")\n    clf.fit(X_train, Y_train)\n    print(\"Training complete!\")\n    \n    # 4. Save artifacts\n    print(f\"Saving model to {MODEL_FILE}...\")\n    joblib.dump(clf, MODEL_FILE, compress=3) # Compressed to save disk space\n    \n    print(f\"Saving metadata to {METADATA_FILE}...\")\n    # Saving vocabulary dictionary guarantees your arrays map perfectly during deployment\n    metadata = {\n        \"selected_go_terms\": selected_go_terms,\n        \"three_mer_to_idx\": THREE_MER_TO_IDX\n    }\n    joblib.dump(metadata, METADATA_FILE)\n    \n    print(\"All training artifacts saved securely and ready for standalone inference!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-07-18T07:11:13.897828Z","iopub.execute_input":"2026-07-18T07:11:13.898156Z","iopub.status.idle":"2026-07-18T07:11:37.745113Z","shell.execute_reply.started":"2026-07-18T07:11:13.898128Z","shell.execute_reply":"2026-07-18T07:11:37.744068Z"}},"outputs":[],"execution_count":null}]}