{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":41875,"databundleVersionId":5521661,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":15667414,"datasetId":10032015,"databundleVersionId":16604439}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-11T12:53:12.741689Z","iopub.execute_input":"2026-04-11T12:53:12.742375Z","iopub.status.idle":"2026-04-11T12:53:12.767575Z","shell.execute_reply.started":"2026-04-11T12:53:12.742341Z","shell.execute_reply":"2026-04-11T12:53:12.76628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install biopython pronto --quiet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T12:43:35.739867Z","iopub.execute_input":"2026-04-11T12:43:35.740167Z","iopub.status.idle":"2026-04-11T12:44:27.415504Z","shell.execute_reply.started":"2026-04-11T12:43:35.740143Z","shell.execute_reply":"2026-04-11T12:44:27.414781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom Bio import SeqIO\nimport pronto\n\n# Load TSV files\ntrain_taxonomy1 = pd.read_csv('/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv', sep='\\t')\ntrain_terms1 = pd.read_csv('/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_terms.tsv', sep='\\t')\n\n# Read FASTA file using Biopython\ntrain_sequences = []\nfor record in SeqIO.parse('/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/train_sequences.fasta', 'fasta'):\n    train_sequences.append({'id': record.id, 'seq': str(record.seq)})\ntrain_sequences_df = pd.DataFrame(train_sequences)\n\n# Read OBO file using pronto\ngo_ontology = pronto.Ontology('/kaggle/input/competitions/cafa-5-protein-function-prediction/Train/go-basic.obo')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T12:52:26.914922Z","iopub.execute_input":"2026-04-11T12:52:26.915642Z","iopub.status.idle":"2026-04-11T12:52:36.645508Z","shell.execute_reply.started":"2026-04-11T12:52:26.915584Z","shell.execute_reply":"2026-04-11T12:52:36.644398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/datasets/ayushdhoble/train-dataset/train_df.csv')\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T12:55:08.796722Z","iopub.execute_input":"2026-04-11T12:55:08.797599Z","iopub.status.idle":"2026-04-11T12:55:14.250113Z","shell.execute_reply.started":"2026-04-11T12:55:08.797542Z","shell.execute_reply":"2026-04-11T12:55:14.249301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T12:55:32.77437Z","iopub.execute_input":"2026-04-11T12:55:32.774689Z","iopub.status.idle":"2026-04-11T12:55:33.307053Z","shell.execute_reply.started":"2026-04-11T12:55:32.774663Z","shell.execute_reply":"2026-04-11T12:55:33.305674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate memory usage in MB\nmemory_usage_bytes = train_df.memory_usage(deep=True).sum()\nmemory_usage_mb = memory_usage_bytes / (1024**2)\n\nprint(f\"The size of train_df is: {train_df.shape}\")\nprint(f\"Memory usage: {memory_usage_mb:.2f} MB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T12:55:36.903204Z","iopub.execute_input":"2026-04-11T12:55:36.903526Z","iopub.status.idle":"2026-04-11T12:55:37.147679Z","shell.execute_reply.started":"2026-04-11T12:55:36.903497Z","shell.execute_reply":"2026-04-11T12:55:37.146889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom Bio.SeqUtils.ProtParam import ProteinAnalysis\nfrom sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\nimport networkx as nx\nimport re","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T12:55:59.732063Z","iopub.execute_input":"2026-04-11T12:55:59.732361Z","iopub.status.idle":"2026-04-11T12:56:01.296108Z","shell.execute_reply.started":"2026-04-11T12:55:59.732312Z","shell.execute_reply":"2026-04-11T12:56:01.295032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Sampling proteins...\")\ndf_sample = train_df.copy()\ndf_sample = df_sample.rename(columns={'id': 'EntryID'})\n\nprint(\"Engineering biochemical features...\")\n\ndef extract_to_dict(seq):\n    try:\n        s = str(seq).upper().strip()\n        clean_seq = re.sub(r'[^ACDEFGHIKLMNPQRSTVWY]', '', s)\n        if len(clean_seq) < 10: return None\n\n        pa = ProteinAnalysis(clean_seq)\n        aa_perc = {}\n        if hasattr(pa, 'amino_acids_percent'):\n            aa_perc = pa.amino_acids_percent\n        elif hasattr(pa, 'get_amino_acids_percent'):\n            aa_perc = pa.get_amino_acids_percent()\n\n        res = {\n            'molecular_weight': pa.molecular_weight(),\n            'aromaticity': pa.aromaticity(),\n            'isoelectric_point': pa.isoelectric_point(),\n            'instability_index': pa.instability_index()\n        }\n        res.update(aa_perc)\n        return res\n    except Exception as e:\n        return None\n\nresults = []\nvalid_indices = []\n\nfor idx, row in df_sample.iterrows():\n    feat = extract_to_dict(row['seq'])\n    if feat is not None:\n        results.append(feat)\n        valid_indices.append(idx)\n\nif results:\n    feat_df = pd.DataFrame(results, index=valid_indices)\n    df_sample = pd.concat([df_sample.loc[valid_indices], feat_df], axis=1)\n\n    # 1. Merge Taxonomy\n    if 'train_taxonomy1' in globals():\n        df_sample = df_sample.merge(train_taxonomy1, on='EntryID', how='left')\n\n    # 2. Enrich with GO Terms and Aspects\n    if 'train_terms1' in globals() and 'go_ontology' in globals():\n        # Get labels for these specific proteins\n        labels = train_terms1[train_terms1['EntryID'].isin(df_sample['EntryID'])]\n\n        # Map GO IDs to names from ontology\n        def get_go_info(go_id):\n            if go_id in go_ontology:\n                term = go_ontology[go_id]\n                return term.name, term.namespace\n            return None, None\n\n        # We'll take the most frequent term per protein for display purposes\n        # or just count them\n        term_counts = labels.groupby('EntryID')['term'].count().reset_index().rename(columns={'term': 'go_label_count'})\n        df_sample = df_sample.merge(term_counts, on='EntryID', how='left').fillna({'go_label_count': 0})\n\n        # Add sample GO information (taking the first associated term as an example metadata point)\n        example_terms = labels.groupby('EntryID').first().reset_index()\n        example_terms[['go_name', 'go_namespace']] = example_terms['term'].apply(lambda x: pd.Series(get_go_info(x)))\n        df_sample = df_sample.merge(example_terms[['EntryID', 'term', 'go_name', 'go_namespace']], on='EntryID', how='left')\n\n    df_sample = df_sample.dropna(axis=0)\n    print(f\"Success! Valid rows after cleaning: {len(df_sample)}\")\n    display(df_sample.head())\nelse:\n    print(\"Extraction failed.\")\n\nprint(\"Size : \", df_sample.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T12:57:43.905681Z","iopub.execute_input":"2026-04-11T12:57:43.906233Z","iopub.status.idle":"2026-04-11T13:00:01.839868Z","shell.execute_reply.started":"2026-04-11T12:57:43.906208Z","shell.execute_reply":"2026-04-11T13:00:01.838524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_sample.to_csv('FE_train.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T13:02:33.785665Z","iopub.execute_input":"2026-04-11T13:02:33.785957Z","iopub.status.idle":"2026-04-11T13:02:39.126996Z","shell.execute_reply.started":"2026-04-11T13:02:33.785935Z","shell.execute_reply":"2026-04-11T13:02:39.125614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!kaggle datasets download -d sergeifironov/t5embeds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T13:04:19.332161Z","iopub.execute_input":"2026-04-11T13:04:19.332472Z","iopub.status.idle":"2026-04-11T13:05:10.067129Z","shell.execute_reply.started":"2026-04-11T13:04:19.332445Z","shell.execute_reply":"2026-04-11T13:05:10.06599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!unzip /kaggle/working/t5embeds.zip","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T13:06:32.680398Z","iopub.execute_input":"2026-04-11T13:06:32.680722Z","iopub.status.idle":"2026-04-11T13:06:50.810313Z","shell.execute_reply.started":"2026-04-11T13:06:32.680691Z","shell.execute_reply":"2026-04-11T13:06:50.809005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_embed = np.load('/kaggle/working/train_embeds.npy')\ntrain_ids = np.load('/kaggle/working/train_ids.npy')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T13:08:31.296788Z","iopub.execute_input":"2026-04-11T13:08:31.297116Z","iopub.status.idle":"2026-04-11T13:08:31.565948Z","shell.execute_reply.started":"2026-04-11T13:08:31.297092Z","shell.execute_reply":"2026-04-11T13:08:31.564729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"embed_df = pd.DataFrame(train_embed)\nembed_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T13:14:59.778621Z","iopub.execute_input":"2026-04-11T13:14:59.778909Z","iopub.status.idle":"2026-04-11T13:14:59.912625Z","shell.execute_reply.started":"2026-04-11T13:14:59.778889Z","shell.execute_reply":"2026-04-11T13:14:59.911682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"embed_df['id'] = train_ids","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T13:15:22.230554Z","iopub.execute_input":"2026-04-11T13:15:22.230808Z","iopub.status.idle":"2026-04-11T13:15:22.250876Z","shell.execute_reply.started":"2026-04-11T13:15:22.230787Z","shell.execute_reply":"2026-04-11T13:15:22.249784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['id'] = train_df['EntryID']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T13:16:12.572591Z","iopub.execute_input":"2026-04-11T13:16:12.572895Z","iopub.status.idle":"2026-04-11T13:16:12.579234Z","shell.execute_reply.started":"2026-04-11T13:16:12.572869Z","shell.execute_reply":"2026-04-11T13:16:12.577921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_df = pd.merge(train_df, embed_df, on='id', how='inner')\nprint(f\"Merged Data Shape: {model_df.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T13:16:14.723794Z","iopub.execute_input":"2026-04-11T13:16:14.724306Z","iopub.status.idle":"2026-04-11T13:16:18.565527Z","shell.execute_reply.started":"2026-04-11T13:16:14.724281Z","shell.execute_reply":"2026-04-11T13:16:18.564658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}