{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Setup","metadata":{}},{"cell_type":"code","source":"!pip install obonet","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:37.906514Z","iopub.execute_input":"2023-05-31T05:07:37.907141Z","iopub.status.idle":"2023-05-31T05:07:49.530083Z","shell.execute_reply.started":"2023-05-31T05:07:37.907098Z","shell.execute_reply":"2023-05-31T05:07:49.529073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport pickle\n\nfrom Bio import SeqIO\nfrom Bio.SeqUtils.ProtParam import ProteinAnalysis\nfrom collections import Counter\nimport obonet\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-31T05:07:49.532885Z","iopub.execute_input":"2023-05-31T05:07:49.53332Z","iopub.status.idle":"2023-05-31T05:07:50.619393Z","shell.execute_reply.started":"2023-05-31T05:07:49.533276Z","shell.execute_reply":"2023-05-31T05:07:50.618377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pickleIO(obj, src, op=\"r\"):\n    if op==\"w\":\n        with open(src, op + \"b\") as f:\n            pickle.dump(obj, f)\n    elif op==\"r\":\n        with open(src, op + \"b\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    else:\n        print(\"unknown operation\")\n        return obj","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:50.623995Z","iopub.execute_input":"2023-05-31T05:07:50.624495Z","iopub.status.idle":"2023-05-31T05:07:50.629556Z","shell.execute_reply.started":"2023-05-31T05:07:50.624466Z","shell.execute_reply":"2023-05-31T05:07:50.628828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set amino acid mapper\naa_map = {'VAL': 'V', 'PRO': 'P', 'ASN': 'N', 'GLU': 'E', 'ASP': 'D', 'ALA': 'A', 'THR': 'T', 'SER': 'S',\n          'LEU': 'L', 'LYS': 'K', 'GLY': 'G', 'GLN': 'Q', 'ILE': 'I', 'PHE': 'F', 'CYS': 'C', 'TRP': 'W',\n          'ARG': 'R', 'TYR': 'Y', 'HIS': 'H', 'MET': 'M'}\naa_map[\"X\"] = \"X\"\naa_map_encoder = {x:y for x,y in zip(list(aa_map.values()),np.arange(21))}\n\n# set amino acid group mapper\naa_groups = {\n    # Electrically Charged Side Chains - positive\n    \"AAG0\": [\"R\", \"H\", \"K\"],\n    # Electrically Charged Side Chains - negative\n    \"AAG1\": [\"D\", \"E\"],\n    # Polar Uncharged Side Chains\n    \"AAG2\": [\"S\", \"T\", \"N\", \"Q\"],\n    # Hydrophobic Side Chains\n    \"AAG3\": [\"A\", \"V\", \"I\", \"L\", \"M\", \"F\", \"Y\", \"W\"],\n    # Not including any group\n    \"AAG4\": [\"P\", \"G\", \"C\", \"X\"],\n}\naa_groups_encoder = {}\nvalue = 0\nfor i in aa_groups.values():\n    for j in i: aa_groups_encoder[j] = value\n    value += 1\ndef get_amino_acids_group_percent(seq):\n    counter = Counter([aa_groups_encoder[i] for i in seq])\n    norm = sum(counter.values())\n    return {f\"AAG{k}\": v / norm for k, v in counter.items()}\n\n# get amino acid properties\n# https://www.kaggle.com/datasets/alejopaullier/aminoacids-physical-and-chemical-properties\naa_props = pd.read_csv(\"/kaggle/input/aminoacids-physical-and-chemical-properties/aminoacids.csv\").set_index('Letter')\n# set property variable for analysis\nPROPS = ['Molecular Weight', 'Residue Weight', 'pKa1', 'pKb2', 'pKx3', 'pl4', \n         'H', 'VSC', 'P1', 'P2', 'SASA', 'NCISC', 'carbon', 'hydrogen', 'oxygen']\n# remove pKx3 which includes na values\nPROPS.remove(\"pKx3\")\n# remove special case amino acid\naa_props = aa_props.drop([\"O\", \"U\"])\n# impute X amino acid property values with mean of other amino acids\nvalue = aa_props.mean()\nfor i in [\"Name\", \"Abbr\", \"Molecular Formula\", \"Residue Formula\"]:\n    value[i] = \"X\"\naa_props.loc[\"X\"] = value\n# shape check\nprint('Amino Acid properties dataframe. Shape:', aa_props.shape )\n# validation check\nfor i in aa_props.index:\n    if i not in aa_map.values():\n        print(i)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:50.630874Z","iopub.execute_input":"2023-05-31T05:07:50.631366Z","iopub.status.idle":"2023-05-31T05:07:50.679872Z","shell.execute_reply.started":"2023-05-31T05:07:50.631333Z","shell.execute_reply":"2023-05-31T05:07:50.67894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"aa_props","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:50.681031Z","iopub.execute_input":"2023-05-31T05:07:50.681327Z","iopub.status.idle":"2023-05-31T05:07:50.732586Z","shell.execute_reply.started":"2023-05-31T05:07:50.6813Z","shell.execute_reply":"2023-05-31T05:07:50.731577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(aa_map_encoder)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:50.733833Z","iopub.execute_input":"2023-05-31T05:07:50.734144Z","iopub.status.idle":"2023-05-31T05:07:50.738748Z","shell.execute_reply.started":"2023-05-31T05:07:50.734117Z","shell.execute_reply":"2023-05-31T05:07:50.737775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(aa_groups_encoder)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:50.740056Z","iopub.execute_input":"2023-05-31T05:07:50.740632Z","iopub.status.idle":"2023-05-31T05:07:50.749173Z","shell.execute_reply.started":"2023-05-31T05:07:50.740598Z","shell.execute_reply":"2023-05-31T05:07:50.748286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GO annotation data (target)","metadata":{}},{"cell_type":"code","source":"df_term = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\", sep=\"\\t\")\ndf_term = df_term.set_index(\"EntryID\")","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:50.75061Z","iopub.execute_input":"2023-05-31T05:07:50.751064Z","iopub.status.idle":"2023-05-31T05:07:53.979824Z","shell.execute_reply.started":"2023-05-31T05:07:50.751028Z","shell.execute_reply":"2023-05-31T05:07:53.978941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_term.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:53.984536Z","iopub.execute_input":"2023-05-31T05:07:53.984917Z","iopub.status.idle":"2023-05-31T05:07:54.001141Z","shell.execute_reply.started":"2023-05-31T05:07:53.984888Z","shell.execute_reply":"2023-05-31T05:07:53.998538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_term.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:54.002382Z","iopub.execute_input":"2023-05-31T05:07:54.002913Z","iopub.status.idle":"2023-05-31T05:07:54.017202Z","shell.execute_reply.started":"2023-05-31T05:07:54.002866Z","shell.execute_reply":"2023-05-31T05:07:54.016025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### GO Term","metadata":{}},{"cell_type":"code","source":"# basic statistics\nprint(\"the number of unique term :\", df_term[\"term\"].nunique())","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:54.018424Z","iopub.execute_input":"2023-05-31T05:07:54.018932Z","iopub.status.idle":"2023-05-31T05:07:54.529589Z","shell.execute_reply.started":"2023-05-31T05:07:54.018899Z","shell.execute_reply":"2023-05-31T05:07:54.528835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# basic statistics\nprint(\"stats of the number of term :\", df_term.groupby(\"EntryID\")[\"term\"].size().describe())","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:54.530516Z","iopub.execute_input":"2023-05-31T05:07:54.531398Z","iopub.status.idle":"2023-05-31T05:07:55.182611Z","shell.execute_reply.started":"2023-05-31T05:07:54.531351Z","shell.execute_reply":"2023-05-31T05:07:55.1816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.percentile(df_term.groupby(\"EntryID\")[\"term\"].size(), 95)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:55.184268Z","iopub.execute_input":"2023-05-31T05:07:55.184681Z","iopub.status.idle":"2023-05-31T05:07:55.831251Z","shell.execute_reply.started":"2023-05-31T05:07:55.184644Z","shell.execute_reply":"2023-05-31T05:07:55.830207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count ratio analysis\ncount = df_term[\"term\"].value_counts(normalize=True)\ncount.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:55.832761Z","iopub.execute_input":"2023-05-31T05:07:55.833877Z","iopub.status.idle":"2023-05-31T05:07:56.336567Z","shell.execute_reply.started":"2023-05-31T05:07:55.833837Z","shell.execute_reply":"2023-05-31T05:07:56.335395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# top 20 ratio\nsns.barplot(x=count.values[:20], y=count.index[:20])\nplt.title(\"Top 20 ratio\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:56.338135Z","iopub.execute_input":"2023-05-31T05:07:56.339124Z","iopub.status.idle":"2023-05-31T05:07:56.713636Z","shell.execute_reply.started":"2023-05-31T05:07:56.339084Z","shell.execute_reply":"2023-05-31T05:07:56.712685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### GO Category","metadata":{}},{"cell_type":"markdown","source":"### 1. In the view of global","metadata":{}},{"cell_type":"code","source":"# basic statistics\nprint(\"number of unique category :\", df_term[\"aspect\"].nunique())","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:56.715034Z","iopub.execute_input":"2023-05-31T05:07:56.715729Z","iopub.status.idle":"2023-05-31T05:07:57.043621Z","shell.execute_reply.started":"2023-05-31T05:07:56.715674Z","shell.execute_reply":"2023-05-31T05:07:57.042423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# count ratio analysis\ncount = df_term[\"aspect\"].value_counts(normalize=True)\ncount.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:57.045071Z","iopub.execute_input":"2023-05-31T05:07:57.045755Z","iopub.status.idle":"2023-05-31T05:07:57.407546Z","shell.execute_reply.started":"2023-05-31T05:07:57.045714Z","shell.execute_reply":"2023-05-31T05:07:57.406461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.pie(x=count.values, labels=count.index, autopct='%.2f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:57.408947Z","iopub.execute_input":"2023-05-31T05:07:57.409262Z","iopub.status.idle":"2023-05-31T05:07:57.531792Z","shell.execute_reply.started":"2023-05-31T05:07:57.409233Z","shell.execute_reply":"2023-05-31T05:07:57.530481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2. Groupby aspect & Counting term","metadata":{}},{"cell_type":"code","source":"for i in [\"MFO\", \"CCO\", \"MFO\"]:\n    count = df_term.loc[df_term[\"aspect\"] == i, \"term\"].value_counts(normalize=True)\n    # top 20 ratio\n    sns.barplot(x=count.values[:20], y=count.index[:20])\n    plt.title(f\"Top 20 ratio on {i}\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:07:57.533386Z","iopub.execute_input":"2023-05-31T05:07:57.534149Z","iopub.status.idle":"2023-05-31T05:08:00.172742Z","shell.execute_reply.started":"2023-05-31T05:07:57.534109Z","shell.execute_reply":"2023-05-31T05:08:00.171769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. Groupby protein & Counting average ratio","metadata":{}},{"cell_type":"code","source":"df_eda = df_term.groupby([\"EntryID\", \"aspect\"]).size().reset_index(name='count')\ndf_eda[\"pct_count\"] = df_eda[\"count\"] / df_eda.groupby('EntryID')['count'].transform('sum').values","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:00.174257Z","iopub.execute_input":"2023-05-31T05:08:00.174846Z","iopub.status.idle":"2023-05-31T05:08:01.518266Z","shell.execute_reply.started":"2023-05-31T05:08:00.174811Z","shell.execute_reply":"2023-05-31T05:08:01.517424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_eda","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:01.51983Z","iopub.execute_input":"2023-05-31T05:08:01.520238Z","iopub.status.idle":"2023-05-31T05:08:01.533443Z","shell.execute_reply.started":"2023-05-31T05:08:01.520197Z","shell.execute_reply":"2023-05-31T05:08:01.532468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Groupby GO aspect & Get distribution statistics on each protein\ncount = df_eda.groupby(\"aspect\")[\"count\"].describe()\ndisplay(count[\"mean\"] / count[\"mean\"].sum())\ndisplay(count)\ncount = df_eda.groupby(\"aspect\")[\"pct_count\"].describe()\ndisplay(count)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:09:15.948549Z","iopub.execute_input":"2023-05-31T05:09:15.94897Z","iopub.status.idle":"2023-05-31T05:09:16.053045Z","shell.execute_reply.started":"2023-05-31T05:09:15.948929Z","shell.execute_reply":"2023-05-31T05:09:16.052077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Average GO aspect ratio on each proteins\ncount = df_eda.groupby(\"aspect\")[\"pct_count\"].mean()\ndisplay(count)\nplt.bar(x=count.index, height=count.values, color=\"lightseagreen\", edgecolor=\"gray\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:01.648789Z","iopub.execute_input":"2023-05-31T05:08:01.649423Z","iopub.status.idle":"2023-05-31T05:08:01.832204Z","shell.execute_reply.started":"2023-05-31T05:08:01.649384Z","shell.execute_reply":"2023-05-31T05:08:01.831209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Appearance ratio\noutput = {}\nfor i in [\"BPO\", \"CCO\", \"MFO\"]:\n    output[i] = (df_eda[\"aspect\"] == i).sum() / df_eda[\"EntryID\"].nunique()\noutput = pd.Series(output)\ndisplay(output)\nplt.bar(x=output.index, height=output.values, color=\"orange\", edgecolor=\"gray\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:01.833515Z","iopub.execute_input":"2023-05-31T05:08:01.834171Z","iopub.status.idle":"2023-05-31T05:08:02.156995Z","shell.execute_reply.started":"2023-05-31T05:08:01.834132Z","shell.execute_reply":"2023-05-31T05:08:02.155933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# combination appearance\noutput = []\ndf_pivot = pd.pivot_table(df_eda, values='pct_count', index='EntryID', columns='aspect', aggfunc=\"count\").fillna(0.0)\nfor idx, row in df_pivot.iterrows():\n    output.append(\"_\".join(row.index[row != 0].to_list()))\ndf_pivot[\"comb\"] = output","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:02.159036Z","iopub.execute_input":"2023-05-31T05:08:02.159894Z","iopub.status.idle":"2023-05-31T05:08:23.978291Z","shell.execute_reply.started":"2023-05-31T05:08:02.159851Z","shell.execute_reply":"2023-05-31T05:08:23.977453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df_pivot[[\"BPO\", \"CCO\", \"MFO\"]].value_counts(normalize=True))","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:23.979332Z","iopub.execute_input":"2023-05-31T05:08:23.979606Z","iopub.status.idle":"2023-05-31T05:08:24.00492Z","shell.execute_reply.started":"2023-05-31T05:08:23.979583Z","shell.execute_reply":"2023-05-31T05:08:24.004253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df_pivot[\"comb\"].value_counts(normalize=True))","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:24.009681Z","iopub.execute_input":"2023-05-31T05:08:24.010263Z","iopub.status.idle":"2023-05-31T05:08:24.030039Z","shell.execute_reply.started":"2023-05-31T05:08:24.010231Z","shell.execute_reply":"2023-05-31T05:08:24.029044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Summary\n1. Mean of BPO ratio on each proteins is about 70%\n2. In the view of frequency for all GO aspect, they include in about a half of train proteins\n    * In other words, many proteins not include all GO aspect categories evenly","metadata":{}},{"cell_type":"markdown","source":"## Graph structure","metadata":{}},{"cell_type":"code","source":"df_term.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:24.031289Z","iopub.execute_input":"2023-05-31T05:08:24.031755Z","iopub.status.idle":"2023-05-31T05:08:24.044008Z","shell.execute_reply.started":"2023-05-31T05:08:24.031718Z","shell.execute_reply":"2023-05-31T05:08:24.043025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"graph = obonet.read_obo(\"/kaggle/input/cafa-5-protein-function-prediction/Train/go-basic.obo\")","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:24.045277Z","iopub.execute_input":"2023-05-31T05:08:24.045581Z","iopub.status.idle":"2023-05-31T05:08:32.211445Z","shell.execute_reply.started":"2023-05-31T05:08:24.045553Z","shell.execute_reply":"2023-05-31T05:08:32.210562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample of graph\ngraph.nodes[\"GO:0034655\"]","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:32.212535Z","iopub.execute_input":"2023-05-31T05:08:32.212993Z","iopub.status.idle":"2023-05-31T05:08:32.219597Z","shell.execute_reply.started":"2023-05-31T05:08:32.212964Z","shell.execute_reply":"2023-05-31T05:08:32.218605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Taxonomy data","metadata":{}},{"cell_type":"code","source":"df_tax = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv\", sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:32.220838Z","iopub.execute_input":"2023-05-31T05:08:32.221772Z","iopub.status.idle":"2023-05-31T05:08:32.314908Z","shell.execute_reply.started":"2023-05-31T05:08:32.221736Z","shell.execute_reply":"2023-05-31T05:08:32.313828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tax","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:32.318394Z","iopub.execute_input":"2023-05-31T05:08:32.319082Z","iopub.status.idle":"2023-05-31T05:08:32.329922Z","shell.execute_reply.started":"2023-05-31T05:08:32.319048Z","shell.execute_reply":"2023-05-31T05:08:32.328717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## IA weight information","metadata":{}},{"cell_type":"code","source":"with open(\"/kaggle/input/cafa-5-protein-function-prediction/IA.txt\", \"r\") as f:\n    tmp = f.readlines()\n    tmp = [i.rstrip(\"\\n\").split(\"\\t\") for i in tmp]\n    res1, res2 = map(list, zip(*tmp))\n    term_weight = pd.Series({k: v for k, v in zip(res1, res2)}).astype(\"float32\")\n# Get only in train set\nterm_weight = term_weight[term_weight.index.isin(df_term['term'])]","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:32.331049Z","iopub.execute_input":"2023-05-31T05:08:32.331333Z","iopub.status.idle":"2023-05-31T05:08:32.767975Z","shell.execute_reply.started":"2023-05-31T05:08:32.331307Z","shell.execute_reply":"2023-05-31T05:08:32.766995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"term_weight.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:32.769206Z","iopub.execute_input":"2023-05-31T05:08:32.769509Z","iopub.status.idle":"2023-05-31T05:08:32.776371Z","shell.execute_reply.started":"2023-05-31T05:08:32.769481Z","shell.execute_reply":"2023-05-31T05:08:32.775423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"term_weight.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:32.777634Z","iopub.execute_input":"2023-05-31T05:08:32.777956Z","iopub.status.idle":"2023-05-31T05:08:32.791507Z","shell.execute_reply.started":"2023-05-31T05:08:32.777929Z","shell.execute_reply":"2023-05-31T05:08:32.790601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Percentage of zero weight\n(term_weight == 0).mean()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:32.792542Z","iopub.execute_input":"2023-05-31T05:08:32.792847Z","iopub.status.idle":"2023-05-31T05:08:32.79842Z","shell.execute_reply.started":"2023-05-31T05:08:32.79282Z","shell.execute_reply":"2023-05-31T05:08:32.797724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Percentage of zero weight in train set\ndf_term[\"term\"].isin(term_weight.index[(term_weight == 0)]).mean()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:32.799456Z","iopub.execute_input":"2023-05-31T05:08:32.800142Z","iopub.status.idle":"2023-05-31T05:08:33.151181Z","shell.execute_reply.started":"2023-05-31T05:08:32.800115Z","shell.execute_reply":"2023-05-31T05:08:33.149924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* We have about 9% of zero score weight GO term","metadata":{}},{"cell_type":"markdown","source":"## Count Analysis","metadata":{}},{"cell_type":"code","source":"df_term[\"weight\"] = df_term[\"term\"].map(term_weight)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:33.15236Z","iopub.execute_input":"2023-05-31T05:08:33.152666Z","iopub.status.idle":"2023-05-31T05:08:33.640531Z","shell.execute_reply.started":"2023-05-31T05:08:33.152639Z","shell.execute_reply":"2023-05-31T05:08:33.639501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# global\nvc = df_term[\"term\"].value_counts(normalize=True)\noutput = []\nfor threshold in np.arange(0.01, 0.99+1e-3, 0.01):\n    n_comp = np.where(vc.cumsum() >= threshold)[0][0] + 1\n    majority_idx = vc.index[:n_comp]\n    output.append((len(majority_idx) / len(vc)) * 100)\n    if threshold in [0.8, 0.9]:\n        print(f\"Cumalative count {threshold} -> {output[-1]}\")\n        print(\"number of classes :\", n_comp)\n        print(\"min count :\", df_term[\"term\"].value_counts().loc[majority_idx[-1]])\n        print(\"weight mean :\", df_term.loc[df_term[\"term\"].isin(majority_idx), \"weight\"].mean())\n        print(\"weight max :\", df_term.loc[df_term[\"term\"].isin(majority_idx), \"weight\"].max())\nsns.lineplot(x=np.arange(0.01, 0.99+1e-3, 0.01), y=output)\nplt.title(\"GO term ratio to Cumalative GO Term total count\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:33.642069Z","iopub.execute_input":"2023-05-31T05:08:33.642375Z","iopub.status.idle":"2023-05-31T05:08:37.203105Z","shell.execute_reply.started":"2023-05-31T05:08:33.642348Z","shell.execute_reply":"2023-05-31T05:08:37.202131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# BPO\nvc = df_term.loc[df_term[\"aspect\"] == \"BPO\", \"term\"].value_counts(normalize=True)\noutput = []\nfor threshold in np.arange(0.01, 0.99+1e-3, 0.01):\n    n_comp = np.where(vc.cumsum() >= threshold)[0][0] + 1\n    majority_idx = vc.index[:n_comp]\n    output.append((len(majority_idx) / len(vc)) * 100)\n    if threshold in [0.8, 0.9]:\n        print(f\"Cumalative count {threshold} -> {output[-1]}\")\n        print(\"number of classes :\", n_comp)\n        print(\"min count :\", df_term[\"term\"].value_counts().loc[majority_idx[-1]])\n        print(\"weight mean :\", df_term.loc[df_term[\"term\"].isin(majority_idx), \"weight\"].mean())\n        print(\"weight max :\", df_term.loc[df_term[\"term\"].isin(majority_idx), \"weight\"].max())\nsns.lineplot(x=np.arange(0.01, 0.99+1e-3, 0.01), y=output)\nplt.title(\"GO term ratio to Cumalative GO Term total count (only BPO)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:37.204316Z","iopub.execute_input":"2023-05-31T05:08:37.204638Z","iopub.status.idle":"2023-05-31T05:08:40.900008Z","shell.execute_reply.started":"2023-05-31T05:08:37.20461Z","shell.execute_reply":"2023-05-31T05:08:40.897297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# MFO\nvc = df_term.loc[df_term[\"aspect\"] == \"MFO\", \"term\"].value_counts(normalize=True)\noutput = []\nfor threshold in np.arange(0.01, 0.99+1e-3, 0.01):\n    n_comp = np.where(vc.cumsum() >= threshold)[0][0] + 1\n    majority_idx = vc.index[:n_comp]\n    output.append((len(majority_idx) / len(vc)) * 100)\n    if threshold in [0.8, 0.9]:\n        print(f\"Cumalative count {threshold} -> {output[-1]}\")\n        print(\"number of classes :\", n_comp)\n        print(\"min count :\", df_term[\"term\"].value_counts().loc[majority_idx[-1]])\n        print(\"weight mean :\", df_term.loc[df_term[\"term\"].isin(majority_idx), \"weight\"].mean())\n        print(\"weight max :\", df_term.loc[df_term[\"term\"].isin(majority_idx), \"weight\"].max())\nsns.lineplot(x=np.arange(0.01, 0.99+1e-3, 0.01), y=output)\nplt.title(\"GO term ratio to Cumalative GO Term total count (only MFO)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:40.90127Z","iopub.execute_input":"2023-05-31T05:08:40.901985Z","iopub.status.idle":"2023-05-31T05:08:43.777894Z","shell.execute_reply.started":"2023-05-31T05:08:40.901947Z","shell.execute_reply":"2023-05-31T05:08:43.776918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CCO\nvc = df_term.loc[df_term[\"aspect\"] == \"CCO\", \"term\"].value_counts(normalize=True)\noutput = []\nfor threshold in np.arange(0.01, 0.99+1e-3, 0.01):\n    n_comp = np.where(vc.cumsum() >= threshold)[0][0] + 1\n    majority_idx = vc.index[:n_comp]\n    output.append((len(majority_idx) / len(vc)) * 100)\n    if threshold in [0.8, 0.9]:\n        print(f\"Cumalative count {threshold} -> {output[-1]}\")\n        print(\"number of classes :\", n_comp)\n        print(\"min count :\", df_term[\"term\"].value_counts().loc[majority_idx[-1]])\n        print(\"weight mean :\", df_term.loc[df_term[\"term\"].isin(majority_idx), \"weight\"].mean())\n        print(\"weight max :\", df_term.loc[df_term[\"term\"].isin(majority_idx), \"weight\"].max())\nsns.lineplot(x=np.arange(0.01, 0.99+1e-3, 0.01), y=output)\nplt.title(\"GO term ratio to Cumalative GO Term total count (only CCO)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:43.779277Z","iopub.execute_input":"2023-05-31T05:08:43.779797Z","iopub.status.idle":"2023-05-31T05:08:46.873478Z","shell.execute_reply.started":"2023-05-31T05:08:43.779758Z","shell.execute_reply":"2023-05-31T05:08:46.872224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Count Analysis (Removing zero weight)","metadata":{}},{"cell_type":"code","source":"df_term_mod = df_term[df_term[\"weight\"] != 0].reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:46.87482Z","iopub.execute_input":"2023-05-31T05:08:46.87523Z","iopub.status.idle":"2023-05-31T05:08:47.714935Z","shell.execute_reply.started":"2023-05-31T05:08:46.875189Z","shell.execute_reply":"2023-05-31T05:08:47.714084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# global\nvc = df_term_mod[\"term\"].value_counts(normalize=True)\noutput = []\nfor threshold in np.arange(0.01, 0.99+1e-3, 0.01):\n    n_comp = np.where(vc.cumsum() >= threshold)[0][0] + 1\n    majority_idx = vc.index[:n_comp]\n    output.append((len(majority_idx) / len(vc)) * 100)\n    if threshold in [0.8, 0.9]:\n        print(f\"Cumalative count {threshold} -> {output[-1]}\")\n        print(\"number of classes :\", n_comp)\n        print(\"min count :\", df_term_mod[\"term\"].value_counts().loc[majority_idx[-1]])\n        print(\"weight mean :\", df_term_mod.loc[df_term_mod[\"term\"].isin(majority_idx), \"weight\"].mean())\n        print(\"weight max :\", df_term_mod.loc[df_term_mod[\"term\"].isin(majority_idx), \"weight\"].max())\nsns.lineplot(x=np.arange(0.01, 0.99+1e-3, 0.01), y=output)\nplt.title(\"GO term ratio to Cumalative GO Term total count\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:47.716012Z","iopub.execute_input":"2023-05-31T05:08:47.716303Z","iopub.status.idle":"2023-05-31T05:08:50.370576Z","shell.execute_reply.started":"2023-05-31T05:08:47.716277Z","shell.execute_reply":"2023-05-31T05:08:50.369633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# BPO\nvc = df_term_mod.loc[df_term_mod[\"aspect\"] == \"BPO\", \"term\"].value_counts(normalize=True)\noutput = []\nfor threshold in np.arange(0.01, 0.99+1e-3, 0.01):\n    n_comp = np.where(vc.cumsum() >= threshold)[0][0] + 1\n    majority_idx = vc.index[:n_comp]\n    output.append((len(majority_idx) / len(vc)) * 100)\n    if threshold in [0.8, 0.9]:\n        print(f\"Cumalative count {threshold} -> {output[-1]}\")\n        print(\"number of classes :\", n_comp)\n        print(\"min count :\", df_term_mod[\"term\"].value_counts().loc[majority_idx[-1]])\n        print(\"weight mean :\", df_term_mod.loc[df_term_mod[\"term\"].isin(majority_idx), \"weight\"].mean())\n        print(\"weight max :\", df_term_mod.loc[df_term_mod[\"term\"].isin(majority_idx), \"weight\"].max())\nsns.lineplot(x=np.arange(0.01, 0.99+1e-3, 0.01), y=output)\nplt.title(\"GO term ratio to Cumalative GO Term total count (only BPO)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:50.371772Z","iopub.execute_input":"2023-05-31T05:08:50.372046Z","iopub.status.idle":"2023-05-31T05:08:53.308026Z","shell.execute_reply.started":"2023-05-31T05:08:50.372022Z","shell.execute_reply":"2023-05-31T05:08:53.30678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# MFO\nvc = df_term_mod.loc[df_term_mod[\"aspect\"] == \"MFO\", \"term\"].value_counts(normalize=True)\noutput = []\nfor threshold in np.arange(0.01, 0.99+1e-3, 0.01):\n    n_comp = np.where(vc.cumsum() >= threshold)[0][0] + 1\n    majority_idx = vc.index[:n_comp]\n    output.append((len(majority_idx) / len(vc)) * 100)\n    if threshold in [0.8, 0.9]:\n        print(f\"Cumalative count {threshold} -> {output[-1]}\")\n        print(\"number of classes :\", n_comp)\n        print(\"min count :\", df_term_mod[\"term\"].value_counts().loc[majority_idx[-1]])\n        print(\"weight mean :\", df_term_mod.loc[df_term_mod[\"term\"].isin(majority_idx), \"weight\"].mean())\n        print(\"weight max :\", df_term_mod.loc[df_term_mod[\"term\"].isin(majority_idx), \"weight\"].max())\nsns.lineplot(x=np.arange(0.01, 0.99+1e-3, 0.01), y=output)\nplt.title(\"GO term ratio to Cumalative GO Term total count (only MFO)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:53.309789Z","iopub.execute_input":"2023-05-31T05:08:53.310189Z","iopub.status.idle":"2023-05-31T05:08:55.804584Z","shell.execute_reply.started":"2023-05-31T05:08:53.310151Z","shell.execute_reply":"2023-05-31T05:08:55.803547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CCO\nvc = df_term_mod.loc[df_term_mod[\"aspect\"] == \"CCO\", \"term\"].value_counts(normalize=True)\noutput = []\nfor threshold in np.arange(0.01, 0.99+1e-3, 0.01):\n    n_comp = np.where(vc.cumsum() >= threshold)[0][0] + 1\n    majority_idx = vc.index[:n_comp]\n    output.append((len(majority_idx) / len(vc)) * 100)\n    if threshold in [0.8, 0.9]:\n        print(f\"Cumalative count {threshold} -> {output[-1]}\")\n        print(\"number of classes :\", n_comp)\n        print(\"min count :\", df_term_mod[\"term\"].value_counts().loc[majority_idx[-1]])\n        print(\"weight mean :\", df_term_mod.loc[df_term_mod[\"term\"].isin(majority_idx), \"weight\"].mean())\n        print(\"weight max :\", df_term_mod.loc[df_term_mod[\"term\"].isin(majority_idx), \"weight\"].max())\nsns.lineplot(x=np.arange(0.01, 0.99+1e-3, 0.01), y=output)\nplt.title(\"GO term ratio to Cumalative GO Term total count (only CCO)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:55.805785Z","iopub.execute_input":"2023-05-31T05:08:55.806071Z","iopub.status.idle":"2023-05-31T05:08:58.825714Z","shell.execute_reply.started":"2023-05-31T05:08:55.806046Z","shell.execute_reply":"2023-05-31T05:08:58.824994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test taxonomy data","metadata":{}},{"cell_type":"code","source":"pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset-taxon-list.tsv\", encoding=\"ISO-8859-1\", sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-05-31T05:08:58.826655Z","iopub.execute_input":"2023-05-31T05:08:58.827454Z","iopub.status.idle":"2023-05-31T05:08:58.844191Z","shell.execute_reply.started":"2023-05-31T05:08:58.827425Z","shell.execute_reply":"2023-05-31T05:08:58.843225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}