{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-22T06:05:29.655781Z","iopub.execute_input":"2023-07-22T06:05:29.656606Z","iopub.status.idle":"2023-07-22T06:05:29.715779Z","shell.execute_reply.started":"2023-07-22T06:05:29.656561Z","shell.execute_reply":"2023-07-22T06:05:29.714925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n","metadata":{"execution":{"iopub.status.busy":"2023-07-22T06:05:38.624682Z","iopub.execute_input":"2023-07-22T06:05:38.625102Z","iopub.status.idle":"2023-07-22T06:05:48.885648Z","shell.execute_reply.started":"2023-07-22T06:05:38.625046Z","shell.execute_reply":"2023-07-22T06:05:48.884565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\",sep=\"\\t\")\nprint(train_terms.shape)\ntrain_terms.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-22T06:09:04.863308Z","iopub.execute_input":"2023-07-22T06:09:04.864604Z","iopub.status.idle":"2023-07-22T06:09:08.926671Z","shell.execute_reply.started":"2023-07-22T06:09:04.864565Z","shell.execute_reply":"2023-07-22T06:09:08.925442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_protein_ids = np.load('/kaggle/input/4637427/train_ids_esm2_t36_3B_UR50D.npy')\n# print(train_protein_ids.shape)\n# train_protein_ids[:5]","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:07:04.447919Z","iopub.execute_input":"2023-07-21T09:07:04.448286Z","iopub.status.idle":"2023-07-21T09:07:04.510285Z","shell.execute_reply.started":"2023-07-21T09:07:04.448247Z","shell.execute_reply":"2023-07-21T09:07:04.509257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_embeddings = np.load('/kaggle/input/4637427/train_embeds_esm2_t36_3B_UR50D.npy')\n# column_num = train_embeddings.shape[1]\n# train_df = pd.DataFrame(train_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\n# print(train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:07:04.513576Z","iopub.execute_input":"2023-07-21T09:07:04.513923Z","iopub.status.idle":"2023-07-21T09:07:32.023818Z","shell.execute_reply.started":"2023-07-21T09:07:04.513891Z","shell.execute_reply":"2023-07-21T09:07:32.021866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])\ntest_protein_ids = np.load('/kaggle/input/4637427/test_ids_esm2_t36_3B_UR50D.npy')\ntest_protein_ids.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-22T06:09:14.217455Z","iopub.execute_input":"2023-07-22T06:09:14.217828Z","iopub.status.idle":"2023-07-22T06:09:14.28293Z","shell.execute_reply.started":"2023-07-22T06:09:14.217799Z","shell.execute_reply":"2023-07-22T06:09:14.282094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:07:32.087479Z","iopub.execute_input":"2023-07-21T09:07:32.087772Z","iopub.status.idle":"2023-07-21T09:07:32.111764Z","shell.execute_reply.started":"2023-07-21T09:07:32.087745Z","shell.execute_reply":"2023-07-21T09:07:32.110767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_of_labels = 1500\nlabels = train_terms['term'].value_counts().index[:num_of_labels].tolist()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-22T06:09:21.238492Z","iopub.execute_input":"2023-07-22T06:09:21.238868Z","iopub.status.idle":"2023-07-22T06:09:22.287786Z","shell.execute_reply.started":"2023-07-22T06:09:21.23884Z","shell.execute_reply":"2023-07-22T06:09:22.286469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# train_terms_updated = train_terms.loc[train_terms['term'].isin(labels)]","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:07:32.645561Z","iopub.execute_input":"2023-07-21T09:07:32.645853Z","iopub.status.idle":"2023-07-21T09:07:33.295702Z","shell.execute_reply.started":"2023-07-21T09:07:32.645828Z","shell.execute_reply":"2023-07-21T09:07:33.294534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_size = train_protein_ids.shape[0] # len(X)\n# train_labels = np.zeros((train_size ,num_of_labels))\n# series_train_protein_ids = pd.Series(train_protein_ids)\n\n# # Loop through each label\n# for i in range(num_of_labels):\n#     # For each label, fetch the corresponding train_terms data\n#     n_train_terms = train_terms_updated[train_terms_updated['term'] ==  labels[i]]\n    \n#     # Fetch all the unique EntryId aka proteins related to the current label(GO term ID)\n#     label_related_proteins = n_train_terms['EntryID'].unique()\n    \n#     # In the series_train_protein_ids pandas series, if a protein is related\n#     # to the current label, then mark it as 1, else 0.\n#     # Replace the ith column of train_Y with with that pandas series.\n#     train_labels[:,i] =  series_train_protein_ids.isin(label_related_proteins).astype(float)\n\n# labels_df = pd.DataFrame(data = train_labels, columns = labels)\n# labels_df.to_numpy().dump('labels_df.npy')\n# print(labels_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:07:33.298946Z","iopub.execute_input":"2023-07-21T09:07:33.299347Z","iopub.status.idle":"2023-07-21T09:16:19.468305Z","shell.execute_reply.started":"2023-07-21T09:07:33.299314Z","shell.execute_reply":"2023-07-21T09:16:19.46705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# labels_df = pd.read_pickle('/kaggle/working/labels_df.npy')","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:16:19.469772Z","iopub.execute_input":"2023-07-21T09:16:19.470202Z","iopub.status.idle":"2023-07-21T09:16:23.585767Z","shell.execute_reply.started":"2023-07-21T09:16:19.470138Z","shell.execute_reply":"2023-07-21T09:16:23.584595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# labels_df = pd.DataFrame(labels_df)\n# labels_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:16:23.587144Z","iopub.execute_input":"2023-07-21T09:16:23.587647Z","iopub.status.idle":"2023-07-21T09:16:23.620087Z","shell.execute_reply.started":"2023-07-21T09:16:23.587607Z","shell.execute_reply":"2023-07-21T09:16:23.618909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INPUT_SHAPE = [train_df.shape[1]]\n# BATCH_SIZE = 5500\n\n# model = tf.keras.Sequential([\n#     tf.keras.layers.BatchNormalization(input_shape=INPUT_SHAPE),    \n#     tf.keras.layers.Dense(units=1000, activation='relu'),\n#      tf.keras.layers.Dropout(0.5, input_shape=(2,)),\n#     tf.keras.layers.Dense(units=1000, activation='relu'),\n#     tf.keras.layers.Dropout(0.5, input_shape=(2,)),\n#      tf.keras.layers.Dense(units=1000, activation='relu'),\n#      tf.keras.layers.Dropout(0.5, input_shape=(2,)),\n#     tf.keras.layers.Dense(units=1000, activation='relu'),\n#      tf.keras.layers.Dropout(0.5, input_shape=(2,)),\n#     tf.keras.layers.Flatten(),\n#     tf.keras.layers.Dense(units=num_of_labels,activation='sigmoid')\n# ])\n\n\n# # Compile model\n# model.compile(\n#     optimizer=tf.keras.optimizers.Adamax(learning_rate=0.001),\n#     loss='binary_crossentropy',\n#     metrics=['binary_accuracy', tf.keras.metrics.AUC()],\n# )\n# model.save(\"model.h5\")\n\n# history = model.fit(\n#     train_df, labels_df,\n#     batch_size=BATCH_SIZE,\n#     epochs=10\n# )","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:19:08.720941Z","iopub.execute_input":"2023-07-21T09:19:08.721434Z","iopub.status.idle":"2023-07-21T09:29:45.790016Z","shell.execute_reply.started":"2023-07-21T09:19:08.721394Z","shell.execute_reply":"2023-07-21T09:29:45.789194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history_df = pd.DataFrame(history.history)\n# history_df.loc[:, ['loss']].plot(title=\"Cross-entropy\")\n# history_df.loc[:, ['binary_accuracy']].plot(title=\"Accuracy\")\n","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:56:48.160427Z","iopub.execute_input":"2023-07-21T09:56:48.161538Z","iopub.status.idle":"2023-07-21T09:56:48.716749Z","shell.execute_reply.started":"2023-07-21T09:56:48.161494Z","shell.execute_reply":"2023-07-21T09:56:48.715722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_embeddings = np.load('/kaggle/input/4637427/test_embeds_esm2_t36_3B_UR50D.npy')\n\n# # Convert test_embeddings to dataframe\n# column_num = test_embeddings.shape[1]\n# test_df = pd.DataFrame(test_embeddings, columns = [\"Column_\" + str(i) for i in range(1, column_num+1)])\n# print(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:57:06.201319Z","iopub.execute_input":"2023-07-21T09:57:06.201743Z","iopub.status.idle":"2023-07-21T09:57:35.443565Z","shell.execute_reply.started":"2023-07-21T09:57:06.201709Z","shell.execute_reply":"2023-07-21T09:57:35.442704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:16:43.792363Z","iopub.status.idle":"2023-07-21T09:16:43.793129Z","shell.execute_reply.started":"2023-07-21T09:16:43.792574Z","shell.execute_reply":"2023-07-21T09:16:43.792611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predictions =  model.predict(test_df)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:57:55.814262Z","iopub.execute_input":"2023-07-21T09:57:55.814626Z","iopub.status.idle":"2023-07-21T09:58:52.490986Z","shell.execute_reply.started":"2023-07-21T09:57:55.814598Z","shell.execute_reply":"2023-07-21T09:58:52.489859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predictions.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-21T10:06:42.676205Z","iopub.execute_input":"2023-07-21T10:06:42.676587Z","iopub.status.idle":"2023-07-21T10:06:42.683368Z","shell.execute_reply.started":"2023-07-21T10:06:42.676558Z","shell.execute_reply":"2023-07-21T10:06:42.682205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# np.save(\"predictions.npy\",predictions)","metadata":{"execution":{"iopub.status.busy":"2023-07-21T10:05:49.103915Z","iopub.execute_input":"2023-07-21T10:05:49.104388Z","iopub.status.idle":"2023-07-21T10:05:49.925899Z","shell.execute_reply.started":"2023-07-21T10:05:49.104353Z","shell.execute_reply":"2023-07-21T10:05:49.922121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = np.load(\"/kaggle/input/predictions/predictions.npy\")","metadata":{"execution":{"iopub.status.busy":"2023-07-22T06:10:54.004601Z","iopub.execute_input":"2023-07-22T06:10:54.005011Z","iopub.status.idle":"2023-07-22T06:11:02.725246Z","shell.execute_reply.started":"2023-07-22T06:10:54.004979Z","shell.execute_reply":"2023-07-22T06:11:02.724117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-22T06:11:23.955563Z","iopub.execute_input":"2023-07-22T06:11:23.956634Z","iopub.status.idle":"2023-07-22T06:11:23.963011Z","shell.execute_reply.started":"2023-07-22T06:11:23.956599Z","shell.execute_reply":"2023-07-22T06:11:23.961825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reference: https://www.kaggle.com/code/alexandervc/baseline-multilabel-to-multitarget-binary\nl = []\nfor k in list(test_protein_ids):\n    l += [ k] * predictions.shape[1]   \n\ndf_submission['Protein Id'] = l\ndf_submission['GO Term Id'] = labels * predictions.shape[0]\ndf_submission['Prediction'] = predictions.ravel()\ndf_submission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-07-22T06:11:34.47493Z","iopub.execute_input":"2023-07-22T06:11:34.475353Z","iopub.status.idle":"2023-07-22T06:29:25.850273Z","shell.execute_reply.started":"2023-07-22T06:11:34.475319Z","shell.execute_reply":"2023-07-22T06:29:25.846908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission","metadata":{"execution":{"iopub.status.busy":"2023-07-22T06:38:59.964261Z","iopub.execute_input":"2023-07-22T06:38:59.964892Z","iopub.status.idle":"2023-07-22T06:38:59.994798Z","shell.execute_reply.started":"2023-07-22T06:38:59.964844Z","shell.execute_reply":"2023-07-22T06:38:59.993259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lm_df_submission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])\n# test_protein_ids = np.load('/kaggle/input/t5embeds/test_ids.npy')\n# l = []\n# for k in list(test_protein_ids):\n#     l += [ k] * lm_predictions.shape[1]   \n\n# lm_df_submission['Protein Id'] = l\n# lm_df_submission['GO Term Id'] = labels * lm_predictions.shape[0]\n# lm_df_submission['Prediction'] = lm_predictions.ravel()\n# lm_df_submission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:16:43.816708Z","iopub.status.idle":"2023-07-21T09:16:43.817275Z","shell.execute_reply.started":"2023-07-21T09:16:43.817064Z","shell.execute_reply":"2023-07-21T09:16:43.817083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lm_df_submission","metadata":{"execution":{"iopub.status.busy":"2023-07-21T09:16:43.819286Z","iopub.status.idle":"2023-07-21T09:16:43.819829Z","shell.execute_reply.started":"2023-07-21T09:16:43.819606Z","shell.execute_reply":"2023-07-21T09:16:43.819624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}