{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n","metadata":{"papermill":{"duration":0.008308,"end_time":"2023-05-09T08:30:09.105539","exception":false,"start_time":"2023-05-09T08:30:09.097231","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Import the Required Libraries","metadata":{"papermill":{"duration":0.009086,"end_time":"2023-05-09T08:30:09.140473","exception":false,"start_time":"2023-05-09T08:30:09.131387","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Required for progressbar widget\nimport progressbar","metadata":{"papermill":{"duration":9.85331,"end_time":"2023-05-09T08:30:19.002985","exception":false,"start_time":"2023-05-09T08:30:09.149675","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T04:32:57.727244Z","iopub.execute_input":"2023-06-23T04:32:57.727673Z","iopub.status.idle":"2023-06-23T04:33:09.167453Z","shell.execute_reply.started":"2023-06-23T04:32:57.727637Z","shell.execute_reply":"2023-06-23T04:33:09.166176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"TensorFlow v\" + tf.__version__)\nprint(\"Numpy v\" + np.__version__)","metadata":{"papermill":{"duration":0.018272,"end_time":"2023-05-09T08:30:19.030432","exception":false,"start_time":"2023-05-09T08:30:19.01216","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T04:33:09.170232Z","iopub.execute_input":"2023-06-23T04:33:09.171543Z","iopub.status.idle":"2023-06-23T04:33:09.177464Z","shell.execute_reply.started":"2023-06-23T04:33:09.171502Z","shell.execute_reply":"2023-06-23T04:33:09.176204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the Dataset","metadata":{"papermill":{"duration":0.008429,"end_time":"2023-05-09T08:30:19.047756","exception":false,"start_time":"2023-05-09T08:30:19.039327","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"First we will load the file `train_terms.tsv` which contains the list of annotated terms (functions) for the proteins. We will extract the labels aka `GO term ID` and create a label dataframe for the protein embeddings.","metadata":{"papermill":{"duration":0.008388,"end_time":"2023-05-09T08:30:19.065367","exception":false,"start_time":"2023-05-09T08:30:19.056979","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_terms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\",sep=\"\\t\")\nprint(train_terms.shape)","metadata":{"papermill":{"duration":3.69155,"end_time":"2023-05-09T08:30:22.766144","exception":false,"start_time":"2023-05-09T08:30:19.074594","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T04:33:09.179044Z","iopub.execute_input":"2023-06-23T04:33:09.179677Z","iopub.status.idle":"2023-06-23T04:33:13.356533Z","shell.execute_reply.started":"2023-06-23T04:33:09.179644Z","shell.execute_reply":"2023-06-23T04:33:13.355056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`train_terms` dataframe is composed of 3 columns and 5363863 entries. We can see all 3 dimensions of our dataset by printing out the first 5 entries using the following code:","metadata":{"papermill":{"duration":0.008358,"end_time":"2023-05-09T08:30:22.783293","exception":false,"start_time":"2023-05-09T08:30:22.774935","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_terms.head()","metadata":{"papermill":{"duration":0.038607,"end_time":"2023-05-09T08:30:22.830633","exception":false,"start_time":"2023-05-09T08:30:22.792026","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T04:33:13.360332Z","iopub.execute_input":"2023-06-23T04:33:13.360868Z","iopub.status.idle":"2023-06-23T04:33:13.396741Z","shell.execute_reply.started":"2023-06-23T04:33:13.36082Z","shell.execute_reply":"2023-06-23T04:33:13.395403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If we look at the first entry of `train_terms.tsv`, we can see that it contains protein id(`A0A009IHW8`), the GO term(`GO:0008152`) and its aspect(`BPO`). ","metadata":{"papermill":{"duration":0.008764,"end_time":"2023-05-09T08:30:22.848867","exception":false,"start_time":"2023-05-09T08:30:22.840103","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Loading the protein embeddings","metadata":{}},{"cell_type":"markdown","source":"First, we will load the protein ids of the protein embeddings in the train dataset contained in `train_ids.npy` into a numpy array.","metadata":{"papermill":{"duration":0.009256,"end_time":"2023-05-09T08:30:22.867158","exception":false,"start_time":"2023-05-09T08:30:22.857902","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_protein_ids = np.load('/kaggle/input/t5embeds/train_ids.npy')\nprint(train_protein_ids.shape)","metadata":{"papermill":{"duration":0.067806,"end_time":"2023-05-09T08:30:22.944355","exception":false,"start_time":"2023-05-09T08:30:22.876549","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T04:33:13.398731Z","iopub.execute_input":"2023-06-23T04:33:13.399173Z","iopub.status.idle":"2023-06-23T04:33:13.460462Z","shell.execute_reply.started":"2023-06-23T04:33:13.399138Z","shell.execute_reply":"2023-06-23T04:33:13.459473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The `train_protein_ids` array consists of 142246 protein_ids. Let us print out the first 5 entries using the following code:","metadata":{"papermill":{"duration":0.009498,"end_time":"2023-05-09T08:30:22.963291","exception":false,"start_time":"2023-05-09T08:30:22.953793","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_protein_ids[:5]","metadata":{"papermill":{"duration":0.019907,"end_time":"2023-05-09T08:30:22.992625","exception":false,"start_time":"2023-05-09T08:30:22.972718","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T04:33:13.461444Z","iopub.execute_input":"2023-06-23T04:33:13.461819Z","iopub.status.idle":"2023-06-23T04:33:13.46921Z","shell.execute_reply.started":"2023-06-23T04:33:13.461776Z","shell.execute_reply":"2023-06-23T04:33:13.468059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!-- Now, we will load`train_embeds.py` which contains the pre-calculated embeddings of the proteins in the train dataset. with protein_ids (`id`s we loaded previously from the **train_ids.npy**) into a numpy array. This array now contains the precalculated embeddings for the protein_ids( Ids we loaded above from **train_ids.npy**) needed for training. -->\n\nAfter loading the files as numpy arrays, we will convert them into Pandas dataframe.\n\nEach protein embedding is a vector of length 1024. We create the resulting dataframe such that there are 1024 columns to represent the values in each of the 1024 places in the vector.","metadata":{"papermill":{"duration":0.009375,"end_time":"2023-05-09T08:30:23.011402","exception":false,"start_time":"2023-05-09T08:30:23.002027","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_embeddings = np.load('/kaggle/input/t5embeds/train_embeds.npy')\n\n# Now lets convert embeddings numpy array(train_embeddings) into pandas dataframe.\ncolumn_num = train_embeddings.shape[1]\ntrain_df = pd.DataFrame(train_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\nprint(train_df.shape)","metadata":{"papermill":{"duration":9.719957,"end_time":"2023-05-09T08:30:32.741095","exception":false,"start_time":"2023-05-09T08:30:23.021138","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T04:33:13.470836Z","iopub.execute_input":"2023-06-23T04:33:13.471198Z","iopub.status.idle":"2023-06-23T04:33:25.382976Z","shell.execute_reply.started":"2023-06-23T04:33:13.471167Z","shell.execute_reply":"2023-06-23T04:33:25.381793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The `train_df` dataframe which contains the embeddings is composed of 1024 columns and 142246 entries. We can see all 1024 dimensions(results will be truncated since column length is too long)  of our dataset by printing out the first 5 entries using the following code:","metadata":{"papermill":{"duration":0.00918,"end_time":"2023-05-09T08:30:32.760375","exception":false,"start_time":"2023-05-09T08:30:32.751195","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train_df.head()","metadata":{"papermill":{"duration":0.036828,"end_time":"2023-05-09T08:30:32.807222","exception":false,"start_time":"2023-05-09T08:30:32.770394","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T04:33:25.384362Z","iopub.execute_input":"2023-06-23T04:33:25.385148Z","iopub.status.idle":"2023-06-23T04:33:25.414616Z","shell.execute_reply.started":"2023-06-23T04:33:25.385107Z","shell.execute_reply":"2023-06-23T04:33:25.413644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare the dataset","metadata":{"papermill":{"duration":0.009208,"end_time":"2023-05-09T08:30:32.825978","exception":false,"start_time":"2023-05-09T08:30:32.81677","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"First we will extract all the needed labels(`GO term ID`) from `train_terms.tsv` file. There are more than 40,000 labels. In order to simplify our model, we will choose the most frequent 1500 `GO term ID`s as labels.","metadata":{"papermill":{"duration":0.009674,"end_time":"2023-05-09T08:30:32.845065","exception":false,"start_time":"2023-05-09T08:30:32.835391","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Let's plot the most frequent 100 `GO Term ID`s in `train_terms.tsv`.","metadata":{"papermill":{"duration":0.009238,"end_time":"2023-05-09T08:30:32.863785","exception":false,"start_time":"2023-05-09T08:30:32.854547","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"We will now save the first 1500 most frequent GO term Ids into a list.","metadata":{"papermill":{"duration":0.010458,"end_time":"2023-05-09T08:30:34.487707","exception":false,"start_time":"2023-05-09T08:30:34.477249","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Set the limit for label\nnum_of_labels = 1500\n\n# Take value counts in descending order and fetch first 1500 `GO term ID` as labels\nlabels = train_terms['term'].value_counts().index[:num_of_labels].tolist()","metadata":{"papermill":{"duration":0.523976,"end_time":"2023-05-09T08:30:35.021974","exception":false,"start_time":"2023-05-09T08:30:34.497998","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T04:33:25.416332Z","iopub.execute_input":"2023-06-23T04:33:25.416701Z","iopub.status.idle":"2023-06-23T04:33:26.475864Z","shell.execute_reply.started":"2023-06-23T04:33:25.416669Z","shell.execute_reply":"2023-06-23T04:33:26.474557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, we will create a new dataframe by filtering the train terms with the selected `GO Term ID`s.","metadata":{"papermill":{"duration":0.009833,"end_time":"2023-05-09T08:30:35.042088","exception":false,"start_time":"2023-05-09T08:30:35.032255","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Fetch the train_terms data for the relevant labels only\ntrain_terms_updated = train_terms.loc[train_terms['term'].isin(labels)]","metadata":{"papermill":{"duration":0.668657,"end_time":"2023-05-09T08:30:35.720953","exception":false,"start_time":"2023-05-09T08:30:35.052296","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T04:33:26.48018Z","iopub.execute_input":"2023-06-23T04:33:26.480578Z","iopub.status.idle":"2023-06-23T04:33:27.402185Z","shell.execute_reply.started":"2023-06-23T04:33:26.480542Z","shell.execute_reply":"2023-06-23T04:33:27.400955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let us plot the aspect values in the new **train_terms_updated** dataframe using a pie chart.","metadata":{"papermill":{"duration":0.009797,"end_time":"2023-05-09T08:30:35.741328","exception":false,"start_time":"2023-05-09T08:30:35.731531","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"As you can see, majority of the `GO term Id`s have BPO(Biological Process Ontology) as their aspect.","metadata":{"papermill":{"duration":0.016642,"end_time":"2023-05-09T08:30:36.204943","exception":false,"start_time":"2023-05-09T08:30:36.188301","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Since this is a multi label classification problem, in the labels array we will denote the presence or absence of each Go Term Id for a protein id using a 1 or 0.\nFirst, we will create a numpy array `train_labels` of required size for the labels. To update the `train_labels` array with the appropriate values, we will loop through the label list.","metadata":{}},{"cell_type":"code","source":"# Setup progressbar settings.\n# This is strictly for aesthetic.\nbar = progressbar.ProgressBar(maxval=num_of_labels, \\\n    widgets=[progressbar.Bar('=', '[', ']'), ' ', progressbar.Percentage()])\n\n# Create an empty dataframe of required size for storing the labels,\n# i.e, train_size x num_of_labels (142246 x 1500)\ntrain_size = train_protein_ids.shape[0] # len(X)\ntrain_labels = np.zeros((train_size ,num_of_labels))\n\n# Convert from numpy to pandas series for better handling\nseries_train_protein_ids = pd.Series(train_protein_ids)\n\n# Loop through each label\nfor i in range(num_of_labels):\n    # For each label, fetch the corresponding train_terms data\n    n_train_terms = train_terms_updated[train_terms_updated['term'] ==  labels[i]]\n    \n    # Fetch all the unique EntryId aka proteins related to the current label(GO term ID)\n    label_related_proteins = n_train_terms['EntryID'].unique()\n    \n    # In the series_train_protein_ids pandas series, if a protein is related\n    # to the current label, then mark it as 1, else 0.\n    # Replace the ith column of train_Y with with that pandas series.\n    train_labels[:,i] =  series_train_protein_ids.isin(label_related_proteins).astype(float)\n    \n    # Progress bar percentage increase\n    bar.update(i+1)\n\n# Notify the end of progress bar \nbar.finish()\n\n# Convert train_Y numpy into pandas dataframe\nlabels_df = pd.DataFrame(data = train_labels, columns = labels)\nprint(labels_df.shape)","metadata":{"papermill":{"duration":495.729474,"end_time":"2023-05-09T08:38:51.951408","exception":false,"start_time":"2023-05-09T08:30:36.221934","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T04:33:27.404035Z","iopub.execute_input":"2023-06-23T04:33:27.404699Z","iopub.status.idle":"2023-06-23T04:53:43.191938Z","shell.execute_reply.started":"2023-06-23T04:33:27.40465Z","shell.execute_reply":"2023-06-23T04:53:43.190848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The final labels dataframe (`label_df`) is composed of 1500 columns and 142246 entries. We can see all 1500 dimensions(results will be truncated since the number of columns is big) of our dataset by printing out the first 5 entries using the following code:","metadata":{"papermill":{"duration":0.010097,"end_time":"2023-05-09T08:38:51.971947","exception":false,"start_time":"2023-05-09T08:38:51.96185","status":"completed"},"tags":[]}},{"cell_type":"code","source":"labels_df.head()","metadata":{"papermill":{"duration":0.048128,"end_time":"2023-05-09T08:38:52.031041","exception":false,"start_time":"2023-05-09T08:38:51.982913","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T04:53:43.1934Z","iopub.execute_input":"2023-06-23T04:53:43.193727Z","iopub.status.idle":"2023-06-23T04:53:43.231501Z","shell.execute_reply.started":"2023-06-23T04:53:43.193698Z","shell.execute_reply":"2023-06-23T04:53:43.230442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training\n\nNext, we will use Tensorflow to train a Deep Neural Network with the protein embeddings.","metadata":{"papermill":{"duration":0.010523,"end_time":"2023-05-09T08:38:52.052433","exception":false,"start_time":"2023-05-09T08:38:52.04191","status":"completed"},"tags":[]}},{"cell_type":"code","source":"INPUT_SHAPE = [train_df.shape[1]]\nBATCH_SIZE = 5500\n\nmodel = tf.keras.Sequential([\n    tf.keras.layers.BatchNormalization(input_shape=INPUT_SHAPE),    \n    tf.keras.layers.Dense(units=1320, activation='relu'),\n     tf.keras.layers.Dropout(0.455, input_shape=(2,)),\n    tf.keras.layers.Dense(units=1320, activation='relu'),\n    tf.keras.layers.Dropout(0.455, input_shape=(2,)),\n     tf.keras.layers.Dense(units=1320, activation='relu'),\n     tf.keras.layers.Dropout(0.455, input_shape=(2,)),\n    tf.keras.layers.Dense(units=1320, activation='relu'),\n     tf.keras.layers.Dropout(0.455, input_shape=(2,)),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(units=num_of_labels,activation='sigmoid')\n])\n\n\n# Compile model\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adamax(learning_rate=0.001),\n    loss='binary_crossentropy',\n    metrics=['binary_accuracy', tf.keras.metrics.AUC()],\n)\n\nhistory = model.fit(\n    train_df, labels_df,\n    batch_size=BATCH_SIZE,\n    epochs=40\n)","metadata":{"papermill":{"duration":128.96621,"end_time":"2023-05-09T08:41:01.029422","exception":false,"start_time":"2023-05-09T08:38:52.063212","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T04:53:43.233225Z","iopub.execute_input":"2023-06-23T04:53:43.234054Z","iopub.status.idle":"2023-06-23T05:46:11.932403Z","shell.execute_reply.started":"2023-06-23T04:53:43.234016Z","shell.execute_reply":"2023-06-23T05:46:11.931216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot the model's loss and accuracy for each epoch","metadata":{"papermill":{"duration":0.019782,"end_time":"2023-05-09T08:41:01.06997","exception":false,"start_time":"2023-05-09T08:41:01.050188","status":"completed"},"tags":[]}},{"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss']].plot(title=\"Cross-entropy\")\nhistory_df.loc[:, ['binary_accuracy']].plot(title=\"Accuracy\")","metadata":{"papermill":{"duration":0.647806,"end_time":"2023-05-09T08:41:01.737745","exception":false,"start_time":"2023-05-09T08:41:01.089939","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T05:46:11.93447Z","iopub.execute_input":"2023-06-23T05:46:11.934893Z","iopub.status.idle":"2023-06-23T05:46:12.736821Z","shell.execute_reply.started":"2023-06-23T05:46:11.934857Z","shell.execute_reply":"2023-06-23T05:46:12.735837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For submission we will use the protein embeddings of the test data created by [Sergei Fironov](https://www.kaggle.com/sergeifironov) using the Rost Lab's T5 protein language model.","metadata":{"papermill":{"duration":0.02075,"end_time":"2023-05-09T08:41:01.82296","exception":false,"start_time":"2023-05-09T08:41:01.80221","status":"completed"},"tags":[]}},{"cell_type":"code","source":"test_embeddings = np.load('/kaggle/input/t5embeds/test_embeds.npy')\n\n# Convert test_embeddings to dataframe\ncolumn_num = test_embeddings.shape[1]\ntest_df = pd.DataFrame(test_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\nprint(test_df.shape)","metadata":{"papermill":{"duration":10.290827,"end_time":"2023-05-09T08:41:12.134919","exception":false,"start_time":"2023-05-09T08:41:01.844092","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T05:46:12.738648Z","iopub.execute_input":"2023-06-23T05:46:12.739384Z","iopub.status.idle":"2023-06-23T05:46:23.929121Z","shell.execute_reply.started":"2023-06-23T05:46:12.73934Z","shell.execute_reply":"2023-06-23T05:46:23.927947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The `test_df` is composed of 1024 columns and 141865 entries. We can see all 1024 dimensions(results will be truncated since column length is too long) of our dataset by printing out the first 5 entries using the following code:","metadata":{"papermill":{"duration":0.020857,"end_time":"2023-05-09T08:41:12.17776","exception":false,"start_time":"2023-05-09T08:41:12.156903","status":"completed"},"tags":[]}},{"cell_type":"code","source":"test_df.head()","metadata":{"papermill":{"duration":0.050123,"end_time":"2023-05-09T08:41:12.248732","exception":false,"start_time":"2023-05-09T08:41:12.198609","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T05:46:23.932828Z","iopub.execute_input":"2023-06-23T05:46:23.933216Z","iopub.status.idle":"2023-06-23T05:46:23.96085Z","shell.execute_reply.started":"2023-06-23T05:46:23.933181Z","shell.execute_reply":"2023-06-23T05:46:23.95967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will now use the model to make predictions on the test embeddings. ","metadata":{}},{"cell_type":"code","source":"predictions =  model.predict(test_df)","metadata":{"papermill":{"duration":663.907351,"end_time":"2023-05-09T08:52:16.178461","exception":false,"start_time":"2023-05-09T08:41:12.27111","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T05:46:23.962205Z","iopub.execute_input":"2023-06-23T05:46:23.962543Z","iopub.status.idle":"2023-06-23T05:47:29.233429Z","shell.execute_reply.started":"2023-06-23T05:46:23.962513Z","shell.execute_reply":"2023-06-23T05:47:29.232048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/code/alexandervc/baseline-multilabel-to-multitarget-binary\n\ndf_submission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])\ntest_protein_ids = np.load('/kaggle/input/t5embeds/test_ids.npy')\nl = []\nfor k in list(test_protein_ids):\n    l += [ k] * predictions.shape[1]   \n\ndf_submission['Protein Id'] = l\ndf_submission['GO Term Id'] = labels * predictions.shape[0]\ndf_submission['Prediction'] = predictions.ravel()\ndf_submission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-06-23T05:47:29.238414Z","iopub.execute_input":"2023-06-23T05:47:29.238862Z","iopub.status.idle":"2023-06-23T06:05:35.411218Z","shell.execute_reply.started":"2023-06-23T05:47:29.238826Z","shell.execute_reply":"2023-06-23T06:05:35.408876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission","metadata":{"papermill":{"duration":0.063739,"end_time":"2023-05-09T08:52:16.292974","exception":false,"start_time":"2023-05-09T08:52:16.229235","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-23T06:05:35.414544Z","iopub.execute_input":"2023-06-23T06:05:35.415151Z","iopub.status.idle":"2023-06-23T06:05:35.436232Z","shell.execute_reply.started":"2023-06-23T06:05:35.415098Z","shell.execute_reply":"2023-06-23T06:05:35.434913Z"},"trusted":true},"execution_count":null,"outputs":[]}]}