{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":41875,"databundleVersionId":5521661,"sourceType":"competition"},{"sourceId":5499219,"sourceType":"datasetVersion","datasetId":3167603}],"dockerImageVersionId":30732,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-15T21:23:55.275644Z","iopub.execute_input":"2024-07-15T21:23:55.276909Z","iopub.status.idle":"2024-07-15T21:23:55.813685Z","shell.execute_reply.started":"2024-07-15T21:23:55.276857Z","shell.execute_reply":"2024-07-15T21:23:55.812292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:23:55.815907Z","iopub.execute_input":"2024-07-15T21:23:55.816474Z","iopub.status.idle":"2024-07-15T21:23:55.826243Z","shell.execute_reply.started":"2024-07-15T21:23:55.816436Z","shell.execute_reply":"2024-07-15T21:23:55.825091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/cafa-5-protein-function-prediction/sample_submission.tsv',sep='\\t', header=None)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:23:55.827865Z","iopub.execute_input":"2024-07-15T21:23:55.828285Z","iopub.status.idle":"2024-07-15T21:23:56.141401Z","shell.execute_reply.started":"2024-07-15T21:23:55.828242Z","shell.execute_reply":"2024-07-15T21:23:56.139742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:23:56.143069Z","iopub.execute_input":"2024-07-15T21:23:56.143622Z","iopub.status.idle":"2024-07-15T21:23:56.177165Z","shell.execute_reply.started":"2024-07-15T21:23:56.143573Z","shell.execute_reply":"2024-07-15T21:23:56.176014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.tsv',sep='\\t', header=False, index=False)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:23:56.179795Z","iopub.execute_input":"2024-07-15T21:23:56.180171Z","iopub.status.idle":"2024-07-15T21:23:56.686634Z","shell.execute_reply.started":"2024-07-15T21:23:56.180137Z","shell.execute_reply":"2024-07-15T21:23:56.68521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Required for progressbar widget\nimport progressbar","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:23:56.688769Z","iopub.execute_input":"2024-07-15T21:23:56.689646Z","iopub.status.idle":"2024-07-15T21:24:12.469055Z","shell.execute_reply.started":"2024-07-15T21:23:56.689599Z","shell.execute_reply":"2024-07-15T21:24:12.467897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"TensorFlow v\" + tf.__version__)\nprint(\"Numpy v\" + np.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:12.470648Z","iopub.execute_input":"2024-07-15T21:24:12.471453Z","iopub.status.idle":"2024-07-15T21:24:12.477558Z","shell.execute_reply.started":"2024-07-15T21:24:12.471419Z","shell.execute_reply":"2024-07-15T21:24:12.476229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\",sep=\"\\t\")\nprint(train_terms.shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:12.479144Z","iopub.execute_input":"2024-07-15T21:24:12.479536Z","iopub.status.idle":"2024-07-15T21:24:15.998055Z","shell.execute_reply.started":"2024-07-15T21:24:12.479505Z","shell.execute_reply":"2024-07-15T21:24:15.996923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:15.999885Z","iopub.execute_input":"2024-07-15T21:24:16.000359Z","iopub.status.idle":"2024-07-15T21:24:16.012165Z","shell.execute_reply.started":"2024-07-15T21:24:16.000301Z","shell.execute_reply":"2024-07-15T21:24:16.010881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_protein_ids = np.load('/kaggle/input/t5embeds/train_ids.npy')","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:16.013728Z","iopub.execute_input":"2024-07-15T21:24:16.0142Z","iopub.status.idle":"2024-07-15T21:24:16.09623Z","shell.execute_reply.started":"2024-07-15T21:24:16.01416Z","shell.execute_reply":"2024-07-15T21:24:16.095132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_protein_ids[:5]","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:16.097765Z","iopub.execute_input":"2024-07-15T21:24:16.098258Z","iopub.status.idle":"2024-07-15T21:24:16.106227Z","shell.execute_reply.started":"2024-07-15T21:24:16.098215Z","shell.execute_reply":"2024-07-15T21:24:16.105155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_protein_ids.shape","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:16.107721Z","iopub.execute_input":"2024-07-15T21:24:16.108078Z","iopub.status.idle":"2024-07-15T21:24:16.118299Z","shell.execute_reply.started":"2024-07-15T21:24:16.108049Z","shell.execute_reply":"2024-07-15T21:24:16.117162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_embeddings = np.load('/kaggle/input/t5embeds/train_embeds.npy')","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:16.119583Z","iopub.execute_input":"2024-07-15T21:24:16.119974Z","iopub.status.idle":"2024-07-15T21:24:30.607881Z","shell.execute_reply.started":"2024-07-15T21:24:16.119942Z","shell.execute_reply":"2024-07-15T21:24:30.606659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_embeddings.shape","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:30.612079Z","iopub.execute_input":"2024-07-15T21:24:30.612463Z","iopub.status.idle":"2024-07-15T21:24:30.619214Z","shell.execute_reply.started":"2024-07-15T21:24:30.612431Z","shell.execute_reply":"2024-07-15T21:24:30.61812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"column_num = train_embeddings.shape[1]\ntrain_df = pd.DataFrame(train_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\nprint(train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:30.620797Z","iopub.execute_input":"2024-07-15T21:24:30.621246Z","iopub.status.idle":"2024-07-15T21:24:30.634013Z","shell.execute_reply.started":"2024-07-15T21:24:30.621208Z","shell.execute_reply":"2024-07-15T21:24:30.632708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:30.63541Z","iopub.execute_input":"2024-07-15T21:24:30.635758Z","iopub.status.idle":"2024-07-15T21:24:30.671807Z","shell.execute_reply.started":"2024-07-15T21:24:30.635729Z","shell.execute_reply":"2024-07-15T21:24:30.670728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select first 1500 values for plotting\nplot_df = train_terms['term'].value_counts().iloc[:100]\n\nfigure, axis = plt.subplots(1, 1, figsize=(12, 6))\n\nbp = sns.barplot(ax=axis, x=np.array(plot_df.index), y=plot_df.values)\nbp.set_xticklabels(bp.get_xticklabels(), rotation=90, size = 6)\naxis.set_title('Top 100 frequent GO term IDs')\nbp.set_xlabel(\"GO term IDs\", fontsize = 12)\nbp.set_ylabel(\"Count\", fontsize = 12)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:30.673132Z","iopub.execute_input":"2024-07-15T21:24:30.673609Z","iopub.status.idle":"2024-07-15T21:24:32.32926Z","shell.execute_reply.started":"2024-07-15T21:24:30.673569Z","shell.execute_reply":"2024-07-15T21:24:32.328137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the limit for label\nnum_of_labels = 1500\n\n# Take value counts in descending order and fetch first 1500 `GO term ID` as labels\nlabels = train_terms['term'].value_counts().index[:num_of_labels].tolist()","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:32.330878Z","iopub.execute_input":"2024-07-15T21:24:32.331234Z","iopub.status.idle":"2024-07-15T21:24:32.900383Z","shell.execute_reply.started":"2024-07-15T21:24:32.331206Z","shell.execute_reply":"2024-07-15T21:24:32.899042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fetch the train_terms data for the relevant labels only\ntrain_terms_updated = train_terms.loc[train_terms['term'].isin(labels)]","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:32.901714Z","iopub.execute_input":"2024-07-15T21:24:32.902037Z","iopub.status.idle":"2024-07-15T21:24:33.561497Z","shell.execute_reply.started":"2024-07-15T21:24:32.90201Z","shell.execute_reply":"2024-07-15T21:24:33.560419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let us plot the aspect values in the new train_terms_updated dataframe using a pie chart.\n\npie_df = train_terms_updated['aspect'].value_counts()\npalette_color = sns.color_palette('bright')\nplt.pie(pie_df.values, labels=np.array(pie_df.index), colors=palette_color, autopct='%.0f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:33.563007Z","iopub.execute_input":"2024-07-15T21:24:33.563468Z","iopub.status.idle":"2024-07-15T21:24:34.011608Z","shell.execute_reply.started":"2024-07-15T21:24:33.563428Z","shell.execute_reply":"2024-07-15T21:24:34.01006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setup progressbar settings.\n# This is strictly for aesthetic.\nbar = progressbar.ProgressBar(maxval=num_of_labels, \\\n    widgets=[progressbar.Bar('=', '[', ']'), ' ', progressbar.Percentage()])\n\n# Create an empty dataframe of required size for storing the labels,\n# i.e, train_size x num_of_labels (142246 x 1500)\ntrain_size = train_protein_ids.shape[0] # len(X)\ntrain_labels = np.zeros((train_size ,num_of_labels))\n\n# Convert from numpy to pandas series for better handling\nseries_train_protein_ids = pd.Series(train_protein_ids)\n\n# Loop through each label\nfor i in range(num_of_labels):\n    # For each label, fetch the corresponding train_terms data\n    n_train_terms = train_terms_updated[train_terms_updated['term'] ==  labels[i]]\n    \n    # Fetch all the unique EntryId aka proteins related to the current label(GO term ID)\n    label_related_proteins = n_train_terms['EntryID'].unique()\n    \n    # In the series_train_protein_ids pandas series, if a protein is related\n    # to the current label, then mark it as 1, else 0.\n    # Replace the ith column of train_Y with with that pandas series.\n    train_labels[:,i] =  series_train_protein_ids.isin(label_related_proteins).astype(float)\n    \n    # Progress bar percentage increase\n    bar.update(i+1)\n\n# Notify the end of progress bar \nbar.finish()\n\n# Convert train_Y numpy into pandas dataframe\nlabels_df = pd.DataFrame(data = train_labels, columns = labels)\nprint(labels_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:24:34.01387Z","iopub.execute_input":"2024-07-15T21:24:34.014466Z","iopub.status.idle":"2024-07-15T21:33:39.897799Z","shell.execute_reply.started":"2024-07-15T21:24:34.014421Z","shell.execute_reply":"2024-07-15T21:33:39.896701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"labels_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:33:39.89902Z","iopub.execute_input":"2024-07-15T21:33:39.899375Z","iopub.status.idle":"2024-07-15T21:33:39.929428Z","shell.execute_reply.started":"2024-07-15T21:33:39.899326Z","shell.execute_reply":"2024-07-15T21:33:39.928177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_SHAPE = [train_df.shape[1]]\nBATCH_SIZE = 5120\n\nmodel = tf.keras.Sequential([\n    tf.keras.layers.BatchNormalization(input_shape=INPUT_SHAPE),    \n    tf.keras.layers.Dense(units=512, activation='relu'),\n    tf.keras.layers.Dense(units=512, activation='relu'),\n    tf.keras.layers.Dense(units=512, activation='relu'),\n    tf.keras.layers.Dense(units=num_of_labels,activation='sigmoid')\n])\n\n\n# Compile model\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy', tf.keras.metrics.AUC()],\n)\n\nhistory = model.fit(\n    train_df, labels_df,\n    batch_size=BATCH_SIZE,\n    epochs=20\n)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:56:34.821726Z","iopub.execute_input":"2024-07-15T21:56:34.822215Z","iopub.status.idle":"2024-07-15T22:06:15.17465Z","shell.execute_reply.started":"2024-07-15T21:56:34.822179Z","shell.execute_reply":"2024-07-15T22:06:15.173361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:43:29.57844Z","iopub.execute_input":"2024-07-15T21:43:29.578977Z","iopub.status.idle":"2024-07-15T21:46:02.709839Z","shell.execute_reply.started":"2024-07-15T21:43:29.578927Z","shell.execute_reply":"2024-07-15T21:46:02.708693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, ['binary_accuracy']].plot(title=\"Accuracy\")","metadata":{"execution":{"iopub.status.busy":"2024-07-15T22:07:05.218585Z","iopub.execute_input":"2024-07-15T22:07:05.219013Z","iopub.status.idle":"2024-07-15T22:07:05.856776Z","shell.execute_reply.started":"2024-07-15T22:07:05.218979Z","shell.execute_reply":"2024-07-15T22:07:05.855268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_embeddings = np.load('/kaggle/input/t5embeds/test_embeds.npy')\n\n# Convert test_embeddings to dataframe\ncolumn_num = test_embeddings.shape[1]\ntest_df = pd.DataFrame(test_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\nprint(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:46:03.445354Z","iopub.execute_input":"2024-07-15T21:46:03.446367Z","iopub.status.idle":"2024-07-15T21:46:18.257889Z","shell.execute_reply.started":"2024-07-15T21:46:03.446308Z","shell.execute_reply":"2024-07-15T21:46:18.256648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:46:18.259613Z","iopub.execute_input":"2024-07-15T21:46:18.26075Z","iopub.status.idle":"2024-07-15T21:46:18.292213Z","shell.execute_reply.started":"2024-07-15T21:46:18.26068Z","shell.execute_reply":"2024-07-15T21:46:18.290909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sklearn \nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:46:18.294161Z","iopub.execute_input":"2024-07-15T21:46:18.294578Z","iopub.status.idle":"2024-07-15T21:46:18.54423Z","shell.execute_reply.started":"2024-07-15T21:46:18.294545Z","shell.execute_reply":"2024-07-15T21:46:18.543129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions =  model.predict(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:46:18.545621Z","iopub.execute_input":"2024-07-15T21:46:18.545981Z","iopub.status.idle":"2024-07-15T21:46:43.474496Z","shell.execute_reply.started":"2024-07-15T21:46:18.545948Z","shell.execute_reply":"2024-07-15T21:46:43.473245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids1 = np.load('/kaggle/input/t5embeds/test_ids.npy')\n\n# Convert test_embeddings to dataframe\n\nprint(test_ids1.shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:46:43.476439Z","iopub.execute_input":"2024-07-15T21:46:43.476876Z","iopub.status.idle":"2024-07-15T21:46:43.551462Z","shell.execute_reply.started":"2024-07-15T21:46:43.476837Z","shell.execute_reply":"2024-07-15T21:46:43.550265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions =  model.predict(test_df)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:46:43.552822Z","iopub.execute_input":"2024-07-15T21:46:43.553247Z","iopub.status.idle":"2024-07-15T21:47:07.587473Z","shell.execute_reply.started":"2024-07-15T21:46:43.553209Z","shell.execute_reply":"2024-07-15T21:47:07.586377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])\ntest_protein_ids = np.load('/kaggle/input/t5embeds/test_ids.npy')\nl = []\nfor k in list(test_protein_ids):\n    l += [ k] * predictions.shape[1]   \n\ndf_submission['Protein Id'] = l\ndf_submission['GO Term Id'] = labels * predictions.shape[0]\ndf_submission['Prediction'] = predictions.ravel()\ndf_submission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:47:07.589386Z","iopub.execute_input":"2024-07-15T21:47:07.589748Z","iopub.status.idle":"2024-07-15T21:55:53.664784Z","shell.execute_reply.started":"2024-07-15T21:47:07.589718Z","shell.execute_reply":"2024-07-15T21:55:53.662627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission","metadata":{"execution":{"iopub.status.busy":"2024-07-15T21:55:53.667939Z","iopub.execute_input":"2024-07-15T21:55:53.668934Z","iopub.status.idle":"2024-07-15T21:55:53.687921Z","shell.execute_reply.started":"2024-07-15T21:55:53.668883Z","shell.execute_reply":"2024-07-15T21:55:53.686611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}