{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":41875,"databundleVersionId":5521661,"sourceType":"competition"},{"sourceId":13850513,"sourceType":"datasetVersion","datasetId":7460079}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport subprocess\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nsubprocess.run([\"pip\", \"install\", \"goatools\", \"-q\"], check=True)\nimport goatools\nfrom goatools.obo_parser import GODag\nfrom functools import lru_cache\nimport pickle, os, time\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or preÇssing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n\n\nimport os, glob\n\nfor f in glob.glob(\"/kaggle/working/*.tsv\"):\n    os.remove(f)\n\nprint(\"🧹 Archivos .tsv anteriores eliminados\")\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-24T12:13:44.334712Z","iopub.execute_input":"2025-11-24T12:13:44.334948Z","iopub.status.idle":"2025-11-24T12:14:24.82807Z","shell.execute_reply.started":"2025-11-24T12:13:44.334928Z","shell.execute_reply":"2025-11-24T12:14:24.827118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #!/usr/bin/env python3\n# # -*- coding: utf-8 -*-\n# \"\"\"\n# detect_random_experimental_addition.py\n# --------------------------------------\n# Performs ONE random experiment comparing two GAF releases.\n\n# Each run:\n#   - Randomly selects two distinct versions (v_start < v_end)\n#   - Detects new experimental annotations appearing in v_end\n#     that were not present in v_start\n#   - Saves submission_<v_start>_to_<v_end>.tsv in /kaggle/working\n\n# ✅ Designed for Kaggle:\n# Input  -> /kaggle/input/fantasia-output/GAF_EXPLORE_CAFA5/\n# Output -> /kaggle/working/\n# \"\"\"\n\n# import polars as pl\n# from pathlib import Path\n# import random\n\n# # =====================================\n# # CONFIGURATION\n# # =====================================\n# FILES = {\n#     214: '/kaggle/input/fantasia-output/GAF_EXPLORE_CAFA5/goa_uniprot_all.gaf.214/goa_uniprot_all_subset.214.tsv',\n#     215: '/kaggle/input/fantasia-output/GAF_EXPLORE_CAFA5/goa_uniprot_all.gaf.215/goa_uniprot_all_subset.215.tsv',\n#     216: '/kaggle/input/fantasia-output/GAF_EXPLORE_CAFA5/goa_uniprot_all.gaf.216/goa_uniprot_all_subset.216.tsv',\n#     217: '/kaggle/input/fantasia-output/GAF_EXPLORE_CAFA5/goa_uniprot_all.gaf.217/goa_uniprot_all_subset.217.tsv',\n#     218: '/kaggle/input/fantasia-output/GAF_EXPLORE_CAFA5/goa_uniprot_all.gaf.218/goa_uniprot_all_subset.218.tsv',\n#     219: '/kaggle/input/fantasia-output/GAF_EXPLORE_CAFA5/goa_uniprot_all.gaf.219/goa_uniprot_all_subset.219.tsv',\n#     220: '/kaggle/input/fantasia-output/GAF_EXPLORE_CAFA5/goa_uniprot_all.gaf.220/goa_uniprot_all_subset.220.tsv',\n#     221: '/kaggle/input/fantasia-output/GAF_EXPLORE_CAFA5/goa_uniprot_all.gaf.221/goa_uniprot_all_subset.221.tsv',\n#     222: '/kaggle/input/fantasia-output/GAF_EXPLORE_CAFA5/goa_uniprot_all.gaf.222/goa_uniprot_all_subset.222.tsv',\n#     223: '/kaggle/input/fantasia-output/GAF_EXPLORE_CAFA5/goa_uniprot_all.gaf.223/goa_uniprot_all_subset.223.tsv',\n#     224: '/kaggle/input/fantasia-output/GAF_EXPLORE_CAFA5/goa_uniprot_all.gaf.224/goa_uniprot_all_subset.224.tsv',\n#     225: '/kaggle/input/fantasia-output/goa_uniprot_all.gaf.225/goa_uniprot_all_subset.225.tsv',\n#     226: '/kaggle/input/fantasia-output/goa_uniprot_all.gaf.226/goa_uniprot_all_subset.226.tsv',\n# }\n\n# EXPERIMENTAL = {\n#     \"EXP\", \"IDA\", \"IPI\", \"IMP\", \"IGI\", \"IEP\", \"HTP\", \"HDA\", \"HMP\", \"HGI\", \"HEP\", \"IC\", \"TAS\",\"IBA\",\"IEA\"\n# }\n\n# # =====================================\n# # LOAD GAF FUNCTION\n# # =====================================\n# def load_gaf(version, path):\n#     print(f\"📂 Loading version {version} ...\")\n#     df = (\n#         pl.read_csv(path, separator=\"\\t\", has_header=True)\n#           .select([\"protein_id\", \"go_term\", \"evidence_code\", \"qualifier\"])\n#           .with_columns([\n#               pl.col(\"protein_id\").cast(pl.Utf8).str.strip_chars(),\n#               pl.col(\"go_term\").cast(pl.Utf8).str.to_uppercase().str.strip_chars(),\n#               pl.col(\"evidence_code\").cast(pl.Utf8).str.to_uppercase().str.strip_chars(),\n#               pl.col(\"qualifier\").cast(pl.Utf8).str.to_uppercase().str.strip_chars(),\n#           ])\n#           .filter(pl.col(\"evidence_code\").is_in(EXPERIMENTAL))\n#           .filter(~pl.col(\"qualifier\").is_in([\"NOT\", \"!NOT\"]))\n#           .select([\"protein_id\", \"go_term\"])\n#           .unique()\n#     )\n#     return df\n\n# # =====================================\n# # RANDOM SINGLE EXPERIMENT\n# # =====================================\n# versions = sorted(FILES.keys())\n# v_start, v_end = sorted(random.sample(versions, 2))\n\n# print(f\"🎲 Random experiment selected:\")\n# print(f\"   🔹 Start version: {v_start}\")\n# print(f\"   🔹 End version:   {v_end}\")\n# print(f\"   🔹 Path A: {FILES[v_start]}\")\n# print(f\"   🔹 Path B: {FILES[v_end]}\")\n\n# # Load datasets\n# gaf_start = load_gaf(v_start, FILES[v_start])\n# gaf_end = load_gaf(v_end, FILES[v_end])\n\n# # Detect additions (new experimental annotations)\n# added = gaf_end.join(gaf_start, on=[\"protein_id\", \"go_term\"], how=\"anti\")\n# added = added.with_columns(pl.lit(1.0).alias(\"score\"))\n\n# # Save submission\n# out_file = Path(f\"/kaggle/working/submission.tsv\")\n# added.write_csv(out_file, separator=\"\\t\", include_header=False)\n\n# print(f\"\\n✅ Submission created: {out_file}\")\n# print(f\"➕ {added.height:,} new experimental annotations detected.\")\n# print(f\"🏁 Experiment {v_start} → {v_end} completed successfully.\\n\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T18:33:19.55179Z","iopub.execute_input":"2025-11-20T18:33:19.552204Z","iopub.status.idle":"2025-11-20T18:33:19.561322Z","shell.execute_reply.started":"2025-11-20T18:33:19.552169Z","shell.execute_reply":"2025-11-20T18:33:19.560104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\nINPUT_FILE = \"/kaggle/input/fantasia-output/submission.tsv\"\nOUTPUT_FILE = \"submission.tsv\"\n\n# Leer SIN asumir cabecera, pero asignando nombres\ndf = pd.read_csv(\n    INPUT_FILE,\n    sep=\"\\t\",\n    header=None,\n    names=[\"protein\", \"GO_term\", \"confidence\"],\n    usecols=[0, 1, 2],\n)\n\nprint(\"Antes de limpiar:\")\nprint(df.head())\n\n# Quitar filas basura tipo cabecera repetida\ndf = df[df[\"protein\"] != \"protein\"]\n\n# Asegurar que 'confidence' es numérico\ndf[\"confidence\"] = 1\n\n# Filtrar por umbral\n# df = df[df[\"confidence\"] >= 0.6]\n\nprint(\"\\nDespués de filtrar:\")\nprint(df.head())\n\n# Guardar SIN cabecera (requisito de la competición)\ndf.to_csv(OUTPUT_FILE, sep=\"\\t\", header=False, index=False)\n\nprint(f\"\\n✅ Archivo '{OUTPUT_FILE}' generado correctamente.\")\nprint(f\"Total de filas: {len(df)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T12:15:14.858811Z","iopub.execute_input":"2025-11-24T12:15:14.859534Z","iopub.status.idle":"2025-11-24T12:15:37.797499Z","shell.execute_reply.started":"2025-11-24T12:15:14.8595Z","shell.execute_reply":"2025-11-24T12:15:37.79665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Estadísticos descriptivos del score\nprint(\"\\n📊 Estadísticos del score:\")\nprint(df['confidence'].describe())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-24T12:15:48.94479Z","iopub.execute_input":"2025-11-24T12:15:48.945082Z","iopub.status.idle":"2025-11-24T12:15:49.289827Z","shell.execute_reply.started":"2025-11-24T12:15:48.945059Z","shell.execute_reply":"2025-11-24T12:15:49.28888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"percentiles = df['confidence'].quantile([0.25, 0.5, 0.75, 0.90, 0.95, 0.99])\nprint(\"\\n📊 Percentiles del score:\")\nprint(percentiles)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-20T18:34:15.667835Z","iopub.execute_input":"2025-11-20T18:34:15.668118Z","iopub.status.idle":"2025-11-20T18:34:16.082549Z","shell.execute_reply.started":"2025-11-20T18:34:15.668096Z","shell.execute_reply":"2025-11-20T18:34:16.081765Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}