{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# for progressbar widget\nimport progressbar\n\n# to help with reproducibilty\nimport random as rn\n\nnp.random.seed(111)\nrn.seed(123)\n\ntf.random.set_seed(99)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-27T14:25:55.608002Z","iopub.execute_input":"2023-07-27T14:25:55.60855Z","iopub.status.idle":"2023-07-27T14:26:05.615334Z","shell.execute_reply.started":"2023-07-27T14:25:55.608466Z","shell.execute_reply":"2023-07-27T14:26:05.614451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check versions\nprint(\"TensorFlow v\" + tf.__version__)\nprint(\"Numpy v\" + np.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:05.617464Z","iopub.execute_input":"2023-07-27T14:26:05.618117Z","iopub.status.idle":"2023-07-27T14:26:05.623373Z","shell.execute_reply.started":"2023-07-27T14:26:05.618081Z","shell.execute_reply":"2023-07-27T14:26:05.622272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load Dataset\ntrain_terms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\",sep=\"\\t\")\nprint(train_terms.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:05.624902Z","iopub.execute_input":"2023-07-27T14:26:05.625995Z","iopub.status.idle":"2023-07-27T14:26:09.08553Z","shell.execute_reply.started":"2023-07-27T14:26:05.625959Z","shell.execute_reply":"2023-07-27T14:26:09.084286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:09.087109Z","iopub.execute_input":"2023-07-27T14:26:09.087739Z","iopub.status.idle":"2023-07-27T14:26:09.110991Z","shell.execute_reply.started":"2023-07-27T14:26:09.087698Z","shell.execute_reply":"2023-07-27T14:26:09.109588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the pre-calculated protein embeddings as a np array\ntrain_protein_ids = np.load('/kaggle/input/t5embeds/train_ids.npy')\nprint(train_protein_ids.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:09.114015Z","iopub.execute_input":"2023-07-27T14:26:09.114346Z","iopub.status.idle":"2023-07-27T14:26:09.172024Z","shell.execute_reply.started":"2023-07-27T14:26:09.114317Z","shell.execute_reply":"2023-07-27T14:26:09.170887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_protein_ids[:5]","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:09.173441Z","iopub.execute_input":"2023-07-27T14:26:09.17378Z","iopub.status.idle":"2023-07-27T14:26:09.181039Z","shell.execute_reply.started":"2023-07-27T14:26:09.17375Z","shell.execute_reply":"2023-07-27T14:26:09.179996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_embeddings = np.load('/kaggle/input/t5embeds/train_embeds.npy')\ntrain_embeddings.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:09.182573Z","iopub.execute_input":"2023-07-27T14:26:09.183298Z","iopub.status.idle":"2023-07-27T14:26:21.016393Z","shell.execute_reply.started":"2023-07-27T14:26:09.183256Z","shell.execute_reply":"2023-07-27T14:26:21.015358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert embeddings np array (train_embeddings) into a pd dataframe\ncolumn_num = train_embeddings.shape[1]\ntrain_df = pd.DataFrame(train_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num + 1)])\nprint(train_df.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:21.017476Z","iopub.execute_input":"2023-07-27T14:26:21.017821Z","iopub.status.idle":"2023-07-27T14:26:21.025251Z","shell.execute_reply.started":"2023-07-27T14:26:21.017793Z","shell.execute_reply":"2023-07-27T14:26:21.023952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df contains the embeddings\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:21.027073Z","iopub.execute_input":"2023-07-27T14:26:21.027639Z","iopub.status.idle":"2023-07-27T14:26:21.0617Z","shell.execute_reply.started":"2023-07-27T14:26:21.027598Z","shell.execute_reply":"2023-07-27T14:26:21.0606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prepare the dataset\n\n# Select first 1500 values for plotting (for simplication since there are more than 4,000 lables)\nplot_df = train_terms['term'].value_counts().iloc[:100]\n\nfigure, axis = plt.subplots(1, 1, figsize = (12, 6))\n\nbp = sns.barplot(ax=axis, x=np.array(plot_df.index), y=plot_df.values)\nbp.set_xticklabels(bp.get_xticklabels(), rotation=90, size = 6)\naxis.set_title('Top 100 frequent GO term IDs')\nbp.set_xlabel(\"GO term IDs\", fontsize = 12)\nbp.set_ylabel(\"Count\", fontsize = 12)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:21.063231Z","iopub.execute_input":"2023-07-27T14:26:21.063927Z","iopub.status.idle":"2023-07-27T14:26:22.938146Z","shell.execute_reply.started":"2023-07-27T14:26:21.063886Z","shell.execute_reply":"2023-07-27T14:26:22.937033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#help(sns.barplot)\n#help(plt.subplots)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:22.939627Z","iopub.execute_input":"2023-07-27T14:26:22.94005Z","iopub.status.idle":"2023-07-27T14:26:22.945515Z","shell.execute_reply.started":"2023-07-27T14:26:22.94001Z","shell.execute_reply":"2023-07-27T14:26:22.944379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set the limit for label\nnum_of_labels = 1500\n\n# Take value counts in descending order and fetch first 1500 'GO term ID' as labels\nlabels = train_terms['term'].value_counts().index[:num_of_labels].tolist()\n#labels","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:22.946756Z","iopub.execute_input":"2023-07-27T14:26:22.947102Z","iopub.status.idle":"2023-07-27T14:26:23.477396Z","shell.execute_reply.started":"2023-07-27T14:26:22.947075Z","shell.execute_reply":"2023-07-27T14:26:23.47628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make a new datafram by filtering the train terms with selected 'GO Term ID'\n\n# Fetch the train_terms data for the relevant labels only\ntrain_terms_updated = train_terms.loc[train_terms['term'].isin(labels)]\n","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:23.478638Z","iopub.execute_input":"2023-07-27T14:26:23.47948Z","iopub.status.idle":"2023-07-27T14:26:24.126826Z","shell.execute_reply.started":"2023-07-27T14:26:23.479449Z","shell.execute_reply":"2023-07-27T14:26:24.12598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms_updated.head","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:24.130272Z","iopub.execute_input":"2023-07-27T14:26:24.131144Z","iopub.status.idle":"2023-07-27T14:26:24.142486Z","shell.execute_reply.started":"2023-07-27T14:26:24.1311Z","shell.execute_reply":"2023-07-27T14:26:24.1413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the aspect values in the new train_terms_updated datafram using a pie chart\n\npie_df = train_terms_updated['aspect'].value_counts()\npalette_color = sns.color_palette('bright')\nplt.pie(pie_df.values, labels=np.array(pie_df.index), colors=palette_color, autopct='%.0f%%')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:24.143684Z","iopub.execute_input":"2023-07-27T14:26:24.144006Z","iopub.status.idle":"2023-07-27T14:26:24.649646Z","shell.execute_reply.started":"2023-07-27T14:26:24.143963Z","shell.execute_reply":"2023-07-27T14:26:24.648307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms_updated.loc[:, 'has_label'] = 1\ndf_pivot = train_terms_updated[train_terms_updated.term.isin(labels)].pivot('EntryID', 'term', 'has_label').fillna(0).astype(int)\nlabels_df = df_pivot.loc[train_protein_ids, labels]","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:24.651552Z","iopub.execute_input":"2023-07-27T14:26:24.65201Z","iopub.status.idle":"2023-07-27T14:26:32.776786Z","shell.execute_reply.started":"2023-07-27T14:26:24.651972Z","shell.execute_reply":"2023-07-27T14:26:32.775578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setup progressbar settings.\n# This is strictly for aesthetic.\nbar = progressbar.ProgressBar(maxval=num_of_labels, \\\n    widgets=[progressbar.Bar('=', '[', ']'), ' ', progressbar.Percentage()])\n\n# Create an empty dataframe of required size for storing the labels,\n# i.e, train_size x num_of_labels (142246 x 1500)\ntrain_size = train_protein_ids.shape[0] # len(X)\ntrain_labels = np.zeros((train_size ,num_of_labels))\n\n# Convert from numpy to pandas series for better handling\nseries_train_protein_ids = pd.Series(train_protein_ids)\n\n# Loop through each label\nfor i in range(num_of_labels):\n    # For each label, fetch the corresponding train_terms data\n    n_train_terms = train_terms_updated[train_terms_updated['term'] ==  labels[i]]\n    \n    # Fetch all the unique EntryId aka proteins related to the current label(GO term ID)\n    label_related_proteins = n_train_terms['EntryID'].unique()\n    \n    # In the series_train_protein_ids pandas series, if a protein is related\n    # to the current label, then mark it as 1, else 0.\n    # Replace the ith column of train_Y with with that pandas series.\n    train_labels[:,i] =  series_train_protein_ids.isin(label_related_proteins).astype(float)\n    \n    # Progress bar percentage increase\n    bar.update(i+1)\n\n# Notify the end of progress bar \nbar.finish()\n\n# Convert train_Y numpy into pandas dataframe\nlabels_df = pd.DataFrame(data = train_labels, columns = labels)\nprint(labels_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:26:32.77812Z","iopub.execute_input":"2023-07-27T14:26:32.778471Z","iopub.status.idle":"2023-07-27T14:35:20.56174Z","shell.execute_reply.started":"2023-07-27T14:26:32.77843Z","shell.execute_reply":"2023-07-27T14:35:20.560523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:35:20.563238Z","iopub.execute_input":"2023-07-27T14:35:20.563601Z","iopub.status.idle":"2023-07-27T14:35:20.59478Z","shell.execute_reply.started":"2023-07-27T14:35:20.563569Z","shell.execute_reply":"2023-07-27T14:35:20.593418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# so i dont have to run above code again\noriginal_trained = train_df","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = original_trained","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Early stopping to prevent crashing\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, callbacks\nearly_stopping = callbacks.EarlyStopping(\n    min_delta = 0.005,\n    patience = 5,\n    restore_best_weights = True,\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:35:20.596005Z","iopub.execute_input":"2023-07-27T14:35:20.596355Z","iopub.status.idle":"2023-07-27T14:35:20.606769Z","shell.execute_reply.started":"2023-07-27T14:35:20.596325Z","shell.execute_reply":"2023-07-27T14:35:20.605562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training (Using Tensorflow to train a Deep Neural Network with the protein \n# embeddings)\n\nINPUT_SHAPE = [train_df.shape[1]]\nBATCH_SIZE = 128\n\nmodel = tf.keras.Sequential([\n    tf.keras.layers.BatchNormalization(input_shape=INPUT_SHAPE), \n    layers.Dropout(rate = 0.3),\n    tf.keras.layers.Dense(units=512, activation='relu'),\n    tf.keras.layers.Dense(units=512, activation='relu'),\n    tf.keras.layers.Dense(units=512, activation='relu'),\n    tf.keras.layers.Dense(units=num_of_labels,activation='sigmoid')\n])\n\n\n# Compile model\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.005),\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy', tf.keras.metrics.AUC()],\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:35:20.608234Z","iopub.execute_input":"2023-07-27T14:35:20.608597Z","iopub.status.idle":"2023-07-27T14:35:20.896219Z","shell.execute_reply.started":"2023-07-27T14:35:20.608566Z","shell.execute_reply":"2023-07-27T14:35:20.894978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_df, labels_df,\n    batch_size=BATCH_SIZE,\n    \n    shuffle = True,\n    \n    validation_split=0.2,\n    \n    epochs= 20,\n    #callbacks = [early_stopping],\n    verbose = 0,\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:35:20.898152Z","iopub.execute_input":"2023-07-27T14:35:20.898617Z","iopub.status.idle":"2023-07-27T14:47:13.050116Z","shell.execute_reply.started":"2023-07-27T14:35:20.898571Z","shell.execute_reply":"2023-07-27T14:47:13.048887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the model's loss and accuracy for each epoch\nhistory_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, ['binary_accuracy']].plot(title=\"Accuracy\")\n","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:47:13.051873Z","iopub.execute_input":"2023-07-27T14:47:13.052251Z","iopub.status.idle":"2023-07-27T14:47:13.717852Z","shell.execute_reply.started":"2023-07-27T14:47:13.052218Z","shell.execute_reply":"2023-07-27T14:47:13.71658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(history.history.keys())","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:47:13.719452Z","iopub.execute_input":"2023-07-27T14:47:13.719993Z","iopub.status.idle":"2023-07-27T14:47:13.726192Z","shell.execute_reply.started":"2023-07-27T14:47:13.719952Z","shell.execute_reply":"2023-07-27T14:47:13.725363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df.loc[:, ['loss', 'val_loss']].plot()","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:47:13.727536Z","iopub.execute_input":"2023-07-27T14:47:13.728122Z","iopub.status.idle":"2023-07-27T14:47:14.016036Z","shell.execute_reply.started":"2023-07-27T14:47:13.728091Z","shell.execute_reply":"2023-07-27T14:47:14.014975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submit the protien emdeddings of the test data usuing the Rost Lab's T% protein language model\ntest_embeddings = np.load('/kaggle/input/t5embeds/test_embeds.npy')\n\n# Convert test_embeddings to dataframe\ncolumn_num = test_embeddings.shape[1]\ntest_df = pd.DataFrame(test_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\nprint(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:47:14.017389Z","iopub.execute_input":"2023-07-27T14:47:14.017753Z","iopub.status.idle":"2023-07-27T14:47:23.739276Z","shell.execute_reply.started":"2023-07-27T14:47:14.017722Z","shell.execute_reply":"2023-07-27T14:47:23.738112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:47:23.742256Z","iopub.execute_input":"2023-07-27T14:47:23.743101Z","iopub.status.idle":"2023-07-27T14:47:23.77232Z","shell.execute_reply.started":"2023-07-27T14:47:23.743061Z","shell.execute_reply":"2023-07-27T14:47:23.771358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Using the model to make predictions of the test\ntest_protein_ids = np.load('/kaggle/input/t5embeds/test_ids.npy')\n\npredictions =  model.predict(test_df)\ndf_submission = pd.DataFrame(predictions, index=test_protein_ids, columns=labels).stack()\ndf_submission = pd.DataFrame({'Protein Id': df_submission.index.get_level_values(0),\n                              'GO Term Id': df_submission.index.get_level_values(1),\n                              'Prediction': df_submission.values\n                             })","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:47:23.773686Z","iopub.execute_input":"2023-07-27T14:47:23.77398Z","iopub.status.idle":"2023-07-27T14:48:22.965162Z","shell.execute_reply.started":"2023-07-27T14:47:23.773954Z","shell.execute_reply":"2023-07-27T14:48:22.964032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions.shape\npredictions[1]","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:48:22.966676Z","iopub.execute_input":"2023-07-27T14:48:22.967021Z","iopub.status.idle":"2023-07-27T14:48:22.975249Z","shell.execute_reply.started":"2023-07-27T14:48:22.966992Z","shell.execute_reply":"2023-07-27T14:48:22.974038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the submission data frame\n# Reference: https://www.kaggle.com/code/alexandervc/baseline-multilabel-to-multitarget-binary\n\ndf_submission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])\n","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:48:22.977477Z","iopub.execute_input":"2023-07-27T14:48:22.977872Z","iopub.status.idle":"2023-07-27T14:48:24.54833Z","shell.execute_reply.started":"2023-07-27T14:48:22.97784Z","shell.execute_reply":"2023-07-27T14:48:24.547188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l = []\nfor k in list(test_protein_ids):\n    l += [ k] * predictions.shape[1]   \n\ndf_submission['Protein Id'] = l\ndf_submission['GO Term Id'] = labels * predictions.shape[0]\ndf_submission['Prediction'] = predictions.ravel()\ndf_submission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-07-27T14:48:24.549784Z","iopub.execute_input":"2023-07-27T14:48:24.550178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}