{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# from warnings import filterwarnings\n# filterwarnings('ignore')\n\n# !pip install -q -U biopython\n# from Bio import SeqIO\n\nimport numpy as np\nimport pandas as pd\n\n# from matplotlib import pyplot as plt\n\nfrom tqdm import tqdm\ntqdm.pandas()\n\n# import h5py\n\n# from sklearn.preprocessing import OneHotEncoder, LabelEncoder\n# from sklearn.model_selection import train_test_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-08T11:51:16.51876Z","iopub.execute_input":"2023-07-08T11:51:16.519207Z","iopub.status.idle":"2023-07-08T11:51:16.52551Z","shell.execute_reply.started":"2023-07-08T11:51:16.519172Z","shell.execute_reply":"2023-07-08T11:51:16.524623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from keras.models import load_model\n# from keras.models import Model\n# from keras.regularizers import l2\n# from keras.callbacks import EarlyStopping\n# from keras.layers import Input, Dense, Dropout, Flatten, Activation\n# from keras.layers import Conv1D, Add, MaxPooling1D, BatchNormalization","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class ProteinsOneHotEncoder:\n    \n#     def __init__(self, id_file, fasta_file= None, max_sequences=0):\n        \n#         self.possible_aa = ['M', 'N', 'S', 'V', 'T', 'H', 'A', 'P', 'Y', 'I', 'D', 'W', 'E','Q', 'L', 'F', 'R', 'K', 'C', 'G', 'U', 'O', 'X', 'B', 'Z']\n#         self.fasta_file = fasta_file\n#         self.max_sequences = max_sequences\n#         self.ids = np.load(id_file)\n#         if self.max_sequences:\n#             self.ids = self.ids[:self.max_sequences]\n#         self.retained_proteins = []\n    \n#     def _fasta_to_dataframe(self):\n        \n#         if self.fasta_file==None:\n#             raise ValueError(\"fasta_file not given\")\n        \n#         records = SeqIO.parse(self.fasta_file, \"fasta\")\n#         data = []\n#         count = 0\n#         for record in tqdm(records, desc='Converting to DataFrame', unit=' Records'):\n#             if self.max_sequences and count >= self.max_sequences:\n#                 break\n#             sequence = str(record.seq)\n#             header = record.id\n#             data.append([header, sequence])\n#             count += 1\n\n#         df = pd.DataFrame(data, columns=[\"Header\", \"Sequence\"])\n#         return df\n\n\n\n#     @staticmethod\n#     def clip(seq : str, length=750):\n#         if (len(seq) > 750):\n#             return seq[: length]\n#         else: return seq\n\n    \n#     @staticmethod\n#     def elongate(seq : str):\n        \n#         n = len(seq)\n\n#         if n > 375:\n#             new_seq = seq + seq[: 750 % n]\n#         elif n > 250:\n#             new_seq = seq*2 + seq[: 750 % n]\n#         elif n > 187.5:\n#             new_seq = seq*3 + seq[: 750 % n]\n#         elif n > 150:\n#             new_seq = seq*4 + seq[: 750 % n]\n#         elif n == 150:\n#             new_seq = seq*5    \n\n#         return new_seq\n\n#     def _clip_and_elongate_sequences(self):\n        \n#         df = self._fasta_to_dataframe()\n#         df['Seq_clip'] = df['Sequence'].progress_apply(self.clip)\n#         self.retained_proteins = [not val for val in df.Seq_clip.progress_apply(str.__len__) < 150]\n#         df = df[self.retained_proteins]\n#         df['Seq_clip'] = df['Seq_clip'].progress_apply(self.elongate)\n#         return df\n    \n#     def encode_data(self):\n\n#         df = self._clip_and_elongate_sequences()\n        \n#         # Create an instance of OneHotEncoder\n#         encoder = OneHotEncoder(sparse_output=False, categories=[self.possible_aa])\n\n#         # Initialize an empty list to store the encoded strings\n#         encoded_strings = []\n\n#         # Iterate over each string in the list\n#         for string in tqdm(df.iloc[:,2].values, desc='Encoding Sequences', unit=' Encodings'):\n#             # Transform the string to a list of individual letters\n#             letters = list(string)\n\n#             # Encode the letters\n#             one_hot_encoded = encoder.fit_transform(np.array(letters).reshape(-1, 1))\n\n#             # Append the encoded string to the list\n#             encoded_strings.append(one_hot_encoded.astype(np.int8))\n\n#         # Convert the list of encoded strings to a numpy array\n#         encoded_array = np.array(encoded_strings)\n        \n#         return encoded_array","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# max_seq = 100000 #@param {type:\"integer\"}\n# train_encoder = ProteinsOneHotEncoder(\n#     id_file = '/kaggle/input/t5embeds/train_ids.npy',\n#     fasta_file = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta',\n#     max_sequences = max_seq\n# )\n# train_encodings = train_encoder.encode_data()\n\n# with h5py.File('/kaggle/input/train-labels-cafa5/train_labels.h5', 'r') as hf:\n#     train_labels = hf['labels'][:]\n\n# if max_seq:\n#     train_labels = train_labels[:max_seq]\n# train_labels = train_labels[train_encoder.retained_proteins]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print('Sanity Check')\n# train_encodings.shape, train_labels.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train, X_test, Y_train, Y_test = train_test_split(train_encodings, train_labels, test_size = 0.1, random_state=123)\n# X_train, X_val, Y_train, Y_val = train_test_split(X_train, Y_train, test_size = 0.11111111111, random_state=123)\n\n# X_train.shape, Y_train.shape, X_val.shape, Y_val.shape, X_test.shape, Y_test.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Utility function: plot model's accuracy and loss\n\n# plt.style.use('seaborn-v0_8-paper')\n\n# def plot_history(history):\n#   binary_accuracy = history.history['binary_accuracy']\n#   val_binary_accuracy = history.history['val_binary_accuracy']\n#   loss = history.history['loss']\n#   val_loss = history.history['val_loss']\n#   x = range(1, len(binary_accuracy) + 1)\n\n#   plt.figure(figsize=(12, 5))\n#   plt.subplot(1, 2, 1)\n#   plt.plot(x[2:], binary_accuracy[2:], 'b', label='Training acc')\n#   plt.plot(x[2:], val_binary_accuracy[2:], 'r', label='Validation acc')\n#   plt.title('Training and validation accuracy')\n#   plt.legend()\n\n#   plt.subplot(1, 2, 2)\n#   plt.plot(x[2:], loss[2:], 'b', label='Training loss')\n#   plt.plot(x[2:], val_loss[2:], 'r', label='Validation loss')\n#   plt.title('Training and validation loss')\n#   plt.legend()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Utility function: Display model score(Loss & Accuracy) across all sets.\n\n# def display_model_score(model, train, val, test, batch_size):\n\n#   train_score = model.evaluate(train[0], train[1], batch_size=batch_size, verbose=1)\n#   print('Train loss: ', train_score[0])\n#   print('Train binary_accuracy: ', train_score[1])\n#   print('-'*70)\n\n#   val_score = model.evaluate(val[0], val[1], batch_size=batch_size, verbose=1)\n#   print('Val loss: ', val_score[0])\n#   print('Val binary_accuracy: ', val_score[1])\n#   print('-'*70)\n  \n#   test_score = model.evaluate(test[0], test[1], batch_size=batch_size, verbose=1)\n#   print('Test loss: ', test_score[0])\n#   print('Test binary_accuracy: ', test_score[1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def residual_block(data, filters, d_rate):\n#   \"\"\"\n#   _data: input\n#   _filters: convolution filters\n#   _d_rate: dilation rate\n#   \"\"\"\n\n#   shortcut = data\n\n#   bn1 = BatchNormalization()(data)\n#   act1 = Activation('relu')(bn1)\n#   conv1 = Conv1D(filters, 1, dilation_rate=d_rate, padding='same', kernel_regularizer=l2(0.001))(act1)\n\n#   #bottleneck convolution\n#   bn2 = BatchNormalization()(conv1)\n#   act2 = Activation('relu')(bn2)\n#   conv2 = Conv1D(filters, 3, padding='same', kernel_regularizer=l2(0.001))(act2)\n\n#   #skip connection\n#   x = Add()([conv2, shortcut])\n\n#   return x","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # model\n\n# x_input = Input(shape=(750, 25))\n\n# #initial conv\n# conv = Conv1D(256, 1, padding='same')(x_input) \n\n# # per-residue representation\n# res1 = residual_block(conv, 256, 2)\n# res2 = residual_block(res1, 256, 3)\n\n# x = MaxPooling1D(3)(res2)\n# x = Dropout(0.5)(x)\n\n# # softmax classifier\n# x = Flatten()(x)\n# x_output = Dense(1500, activation='sigmoid', kernel_regularizer=l2(0.0001))(x)\n\n# model_CNN = Model(inputs=x_input, outputs=x_output)\n# model_CNN.compile(optimizer='adam', loss='binary_crossentropy', metrics=['binary_accuracy'])\n\n# model_CNN.summary()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Early Stopping\n# es = EarlyStopping(monitor='val_loss', patience=3, verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history = model_CNN.fit(\n#     X_train, Y_train,\n#     epochs=10, batch_size=256,\n#     validation_data= (X_val,Y_val),\n#     callbacks=[es]\n#     )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # saving model weights.\n# model_CNN.save_weights('model_CNN.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot_history(history)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display_model_score(\n#     model_CNN,\n#     [X_train, Y_train],\n#     [X_val, Y_val],\n#     [X_test, Y_test],\n#     256)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import hamming_loss\n# pred_tn = np.round(model_CNN.predict(X_train))\n# print(f'Hamming Loss On Training Set = {hamming_loss(Y_train, pred_tn)}')\n# pred_tt = np.round(model_CNN.predict(X_test))\n# print(f'Hamming Loss On Test Set = {hamming_loss(Y_test, pred_tt)}')\n# pred_val = np.round(model_CNN.predict(X_val))\n# print(f'Hamming Loss On Validation Set = {hamming_loss(Y_val, pred_val)}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# max_seq = 0\n# test_encoder = ProteinsOneHotEncoder(\n#     id_file = '/kaggle/input/t5embeds/test_ids.npy',\n#     fasta_file = '/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta',\n#     max_sequences = max_seq\n# )\n# test_encodings = test_encoder.encode_data()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# np.save('test_proteins_retained.npy', test_encoder.retained_proteins)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model_CNN.load_weights('/kaggle/input/train-labels-cafa5/model_CNN.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predictions = model_CNN.predict(test_encodings)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# np.save('predictions_CNN.npy', predictions)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = np.load('/kaggle/input/cafa-5-cnn/predictions_CNN.npy')","metadata":{"execution":{"iopub.status.busy":"2023-07-08T11:51:27.364239Z","iopub.execute_input":"2023-07-08T11:51:27.36468Z","iopub.status.idle":"2023-07-08T11:51:33.170391Z","shell.execute_reply.started":"2023-07-08T11:51:27.364645Z","shell.execute_reply":"2023-07-08T11:51:33.16843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = np.load('/kaggle/input/train-labels-cafa5/top_1500_labels.npy',allow_pickle=True)\nslice_ = np.load('/kaggle/input/train-labels-cafa5/test_proteins_retained.npy')","metadata":{"execution":{"iopub.status.busy":"2023-07-08T11:52:18.074184Z","iopub.execute_input":"2023-07-08T11:52:18.07459Z","iopub.status.idle":"2023-07-08T11:52:18.096259Z","shell.execute_reply.started":"2023-07-08T11:52:18.07456Z","shell.execute_reply":"2023-07-08T11:52:18.094452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids = np.load('/kaggle/input/t5embeds/test_ids.npy')\ntest_protein_ids = test_ids[slice_]","metadata":{"execution":{"iopub.status.busy":"2023-07-08T11:54:18.514632Z","iopub.execute_input":"2023-07-08T11:54:18.515037Z","iopub.status.idle":"2023-07-08T11:54:18.581726Z","shell.execute_reply.started":"2023-07-08T11:54:18.515007Z","shell.execute_reply":"2023-07-08T11:54:18.580108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l = []\nfor k in tqdm(list(test_protein_ids)):\n    l += [ k] * predictions.shape[1]","metadata":{"execution":{"iopub.status.busy":"2023-07-08T11:54:25.758344Z","iopub.execute_input":"2023-07-08T11:54:25.758785Z","iopub.status.idle":"2023-07-08T11:54:28.648709Z","shell.execute_reply.started":"2023-07-08T11:54:25.75874Z","shell.execute_reply":"2023-07-08T11:54:28.647413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.DataFrame(\n    {\n        'Protein ID': l,\n        'GO Term ID': np.tile(labels, predictions.shape[0]),\n        'Prediction': np.round(predictions.ravel(),3)\n    }\n)","metadata":{"execution":{"iopub.status.busy":"2023-07-08T11:54:33.283548Z","iopub.execute_input":"2023-07-08T11:54:33.283938Z","iopub.status.idle":"2023-07-08T11:54:58.396759Z","shell.execute_reply.started":"2023-07-08T11:54:33.28391Z","shell.execute_reply":"2023-07-08T11:54:58.394831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-08T11:55:01.734752Z","iopub.execute_input":"2023-07-08T11:55:01.735167Z","iopub.status.idle":"2023-07-08T11:55:02.538812Z","shell.execute_reply.started":"2023-07-08T11:55:01.735137Z","shell.execute_reply":"2023-07-08T11:55:02.537594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.to_csv('submission.tsv', sep='\\t', header=None, index= None)","metadata":{"execution":{"iopub.status.busy":"2023-07-08T11:55:31.139245Z","iopub.execute_input":"2023-07-08T11:55:31.139746Z","iopub.status.idle":"2023-07-08T11:55:31.155244Z","shell.execute_reply.started":"2023-07-08T11:55:31.139707Z","shell.execute_reply":"2023-07-08T11:55:31.153742Z"},"trusted":true},"execution_count":null,"outputs":[]}]}