{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Draft under construction\n\n# What is about ?\n\nLearn pytorch using CAFA5 as example. \nBased on https://www.kaggle.com/code/andreylalaley/pytorch-cafa-5-prediction , but simplified and updated. \n\n\n#### Background:\n\n\nSimple example of NN on pytorch: https://www.kaggle.com/code/alexandervc/moa-nn-04/notebook\n\nThe most simple exercises to implement Linear and Logistic regression: \nhttps://www.kaggle.com/code/alexandervc/pytorch-03-basics-implement-linear-regressions\nhttps://www.kaggle.com/code/alexandervc/pytorch-04-basics-implement-logregression\n\n\n#### Further steps: \n\n\nLater on try understand/modify/apply to CAFA5: \n\nhttps://www.kaggle.com/code/romanvinogradov/cafa-mlp-inference\n\nIdea by \"tmp\" - use cnn for such types of data, but make a first dense layer which arranges data to be suitable for cnn:  https://www.kaggle.com/competitions/lish-moa/discussion/202256\n\nRelated ideas: Senkin13: https://www.kaggle.com/code/senkin13/2nd-place-gru-cite , modifications by Dmitry Ershov: \nhttps://www.kaggle.com/code/bejeweled/rotate-2nd-place-cite-2d-cnn and other 3 notebooks by him (top-scored on Kaggle Open Problems (late submissions)): https://www.kaggle.com/competitions/open-problems-multimodal/code?competitionId=38128&sortBy=scoreDescending\n","metadata":{}},{"cell_type":"markdown","source":"# Preparations ","metadata":{}},{"cell_type":"code","source":"# (MG)\n# Эта штука красиво рисует структуру сети, практической пользы нет.\n\n# %%capture\n# !pip install torchsummary","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:42:48.188522Z","iopub.execute_input":"2023-08-02T15:42:48.189526Z","iopub.status.idle":"2023-08-02T15:42:48.213672Z","shell.execute_reply.started":"2023-08-02T15:42:48.189451Z","shell.execute_reply":"2023-08-02T15:42:48.212852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# (MG)\n# Перетащил в начало подключение разных библиотек, убрал дубли.\n\n## https://www.kaggle.com/code/alexandervc/baseline-multilabel-to-multitarget-binary#Load-train-features---precalculated-embeddings-for-the-proteins\n\nimport os\nimport gc\n\nimport time\nt0start = time.time()\n\nimport random\n\nfrom sklearn.model_selection import train_test_split\n\nimport numpy as np # linear algebra\n\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport torch\nimport torch.nn as nn\n\nimport seaborn as sns\n\nimport matplotlib.pyplot as plt\n\nfrom tqdm.notebook import tqdm\ntqdm.pandas()\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom sklearn.model_selection import train_test_split\n\nfrom torchmetrics import AUROC,F1Score\nfrom torchmetrics.classification import BinaryF1Score\n\n# import torch.nn.functional as F\n# from torch.utils.data import Dataset, DataLoader\n# from torchsummary import summary as torchsummary","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:42:48.215212Z","iopub.execute_input":"2023-08-02T15:42:48.215683Z","iopub.status.idle":"2023-08-02T15:43:03.030728Z","shell.execute_reply.started":"2023-08-02T15:42:48.215656Z","shell.execute_reply":"2023-08-02T15:43:03.029494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_seed = 123\nprint(\"fRandom seed: {random_seed}\")\nrandom.seed(random_seed)\ntorch.manual_seed(random_seed)\n# torch.use_deterministic_algorithms(True) # Needed for reproducible results","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:03.032326Z","iopub.execute_input":"2023-08-02T15:43:03.032683Z","iopub.status.idle":"2023-08-02T15:43:03.045603Z","shell.execute_reply.started":"2023-08-02T15:43:03.032654Z","shell.execute_reply":"2023-08-02T15:43:03.044347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# (MG)\n# Тут были дубли подключения библиотек.\n\n# import seaborn as sns\n# import matplotlib.pyplot as plt\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# (MG)\n# Позновательно, но побережом машинное время.\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:03.048568Z","iopub.execute_input":"2023-08-02T15:43:03.048928Z","iopub.status.idle":"2023-08-02T15:43:03.055337Z","shell.execute_reply.started":"2023-08-02T15:43:03.0489Z","shell.execute_reply":"2023-08-02T15:43:03.054244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Choose device - GPU or CPU and assign model to it","metadata":{}},{"cell_type":"code","source":"# (MG)\n# Перетащил этот блок снизу сюда, что бы не смешивался с блоками нейронной сети.\n\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Using device: {device}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:03.060404Z","iopub.execute_input":"2023-08-02T15:43:03.060805Z","iopub.status.idle":"2023-08-02T15:43:03.07203Z","shell.execute_reply.started":"2023-08-02T15:43:03.060763Z","shell.execute_reply":"2023-08-02T15:43:03.070899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metrics \n\n\"torchmetrics\" - is additional package to pytorch - contains many metrics which basic pytorch does not \n\nPay attention:  input arguments should be torch.tensor not lists or np.arrays \n\nPay attention:  auc function - order of arguments is opposite to sklearn rocauc function, and on some other details of intialization - different from the sklearn syntax. The syntax is auroc - not auCroc - as one may expect\n\nFunny note: same losses e.g. BCE can be imported from different modelues with differnt names, but they are the same - e.g. torch.nn.functional under the name binary_cross_entropy and import torch.nn as nn - as  nn.BCELoss()","metadata":{}},{"cell_type":"code","source":"\n# (MG)\n# Какой-то эксперимент с метриками.\n\n# import torch.nn.functional as F\n\n# # F.binary_cross_entropy(out, labels)   # Calculate loss\n# s =  F.binary_cross_entropy( torch.tensor([0.,1,0.5]), torch.tensor([0,1.,1] ) )  # Important to have float - on [0,1] it crashes - \"Long not supported\"\n# print( s )\n# print( s.grad_fn )\n\n# import torch.nn as nn\n# preds = torch.tensor([0.,1,0.5])\n# target = torch.tensor([0,1.,1])\n# loss_func = nn.BCELoss()\n# s = loss_func(preds, target )\n# print( s )\n# print( s.grad_fn )\n\n# import torch.nn as nn\n# preds = torch.tensor([0.,1,0.5])\n# target = torch.tensor([0,1.,1])\n# loss_func  = torch.nn.MSELoss() \n# s = loss_func(preds, target )\n# print( s )\n# print( s.grad_fn )\n\n\n# from torchmetrics import AUROC,F1Score\n\n# auroc = AUROC(task = 'binary')\n# # order of arguments is different from the sklearn : sklearn - y_true, y_score, while in pytorch - y_score, y_true\n# # Also we first need to initialize  aucroc = AUROC(task = 'binary'), and only then apply  \n# s = auroc(torch.tensor([0,.6,0]), torch.tensor([0,1,1]) ) \n# print('\\n AUC')\n# print( s )\n# print( s.grad_fn )\n# s = auroc(torch.tensor([0,.01,0.02,0]), torch.tensor([0,1,1,1]) ) \n# print( s )\n# s = auroc(torch.tensor([0,.1,0.2,0]), torch.tensor([0,1,1,1]) ) \n# print( s )\n# print()\n\n# target = torch.tensor([0, 1, 1, 0, 1, 1])\n# preds = torch.tensor([0, 1, 1, 0, 0, 1])\n# f1 = F1Score(task=\"multiclass\", num_classes=2)\n# s = f1(preds, target)\n# print( s )\n# print( s.grad_fn )\n\n# from torchmetrics.classification import BinaryF1Score\n# f1 = BinaryF1Score(threshold=0.5, multidim_average= 'global'  )\n# s = f1(preds, target)\n# print( s )\n# print( s.grad_fn )","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:03.07346Z","iopub.execute_input":"2023-08-02T15:43:03.073763Z","iopub.status.idle":"2023-08-02T15:43:03.084651Z","shell.execute_reply.started":"2023-08-02T15:43:03.073737Z","shell.execute_reply":"2023-08-02T15:43:03.083359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## How multi-target cases are computed \n\nMany  metrics support cases of many tartgets, i.e. Y_pred and Y_true are both matrices : n_samples x n_targets \n\nThere is NO such support in e.g. sklearn.metrics - where we typically work with single target tasks. \n\nThere are several ways how standard metrics can be extended to multi-target cases,\nand for some metrics there is a parameter \"multidim_average\", taking values 'global'  / 'samplewise' which control what way to choose.\n\n\nThe other detail - for the metrics like F1 predictions expected to be in [0,1] interval , if this fail, then torch and sklearn metrics may disagree. In standard case the agree as shown below. \n","metadata":{}},{"cell_type":"code","source":"# (MG)\n# Какой-то эксперимент с метриками.\n\n# target =  torch.tensor( (np.random.randn(100,10)>0).astype(float) )\n# preds = torch.tensor( np.clip( np.random.randn(100,10) + 0.5, 0 ,1) )\n\n# #  multidim_average= 'global'\n# print( \" multidim_average= 'global'   \", 'Scalar will be returned')\n# f1 = BinaryF1Score(threshold=0.5, multidim_average= 'global'  )\n# s = f1(preds, target)\n# print(s.shape)\n# print( s )\n\n# print()\n# print(\" multidim_average= 'samplewise'  \", 'vector of values for each sample will be returned')\n# f1 = BinaryF1Score(threshold=0.5, multidim_average= 'samplewise'  )\n# s = f1(preds, target)\n# print(s.shape)\n# print(s.mean() )\n# print( s )\n\n# print()\n# print()\n# print('Implement the same with sklearn')\n# from sklearn.metrics import f1_score\n# print('Global case ')\n# s = f1_score( target.numpy().ravel() ,   preds.numpy().ravel() >= 0.5  )\n# print(np.round(s,4))\n\n# print('Samplewise case:')\n# l = [ f1_score( target.numpy()[i,:], preds.numpy()[i,:] >= 0.5 ) for i in range(preds.shape[0]) ]\n# s = np.mean(l)\n# print( np.round(s,4))","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:03.086888Z","iopub.execute_input":"2023-08-02T15:43:03.087546Z","iopub.status.idle":"2023-08-02T15:43:03.101077Z","shell.execute_reply.started":"2023-08-02T15:43:03.087507Z","shell.execute_reply":"2023-08-02T15:43:03.099846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# (MG)\n# Какой-то эксперимент с метриками.\n\n# %%time\n# s = auroc(preds, target ) \n# print( s )\n\n\n# from sklearn.metrics import roc_auc_score\n# s = roc_auc_score(target.numpy().ravel() , preds.numpy().ravel()) \n# print( np.round(s,4) )","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:03.103026Z","iopub.execute_input":"2023-08-02T15:43:03.103433Z","iopub.status.idle":"2023-08-02T15:43:03.117002Z","shell.execute_reply.started":"2023-08-02T15:43:03.103397Z","shell.execute_reply":"2023-08-02T15:43:03.116106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## do NOT do like that - canNOT apply metrics  to list/np.arrays etc  - we can  apply metrics to pytorch tensors only","metadata":{}},{"cell_type":"code","source":"# (MG)\n# Какой-то эксперимент с метриками.\n\n# try: \n#     F.binary_cross_entropy( np.array([0.,1.,0.5]), np.array([0.,1.,1.] ) )\n#     F.binary_cross_entropy( [0.,1.,0.5] ,  [0.,1.,1.]  )\n#     aucroc(([0,.6,0]), ([0,1,1]) )\n#     aucroc(np.array([0,.6,0]), np.array([0,1,1]) )\n# except Exception as inst:\n#     print(type(inst))    # the exception type\n#     print(inst.args)     # arguments stored in .args\n#     print(inst)      ","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:03.121211Z","iopub.execute_input":"2023-08-02T15:43:03.121936Z","iopub.status.idle":"2023-08-02T15:43:03.129097Z","shell.execute_reply.started":"2023-08-02T15:43:03.121907Z","shell.execute_reply":"2023-08-02T15:43:03.128356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load X,Y, etc \n\nLoad precalulcated features - embeddings and targets transformed in 0,1 multi-target task.","metadata":{}},{"cell_type":"code","source":"%%time \n\n################ load  features #########################\n\nfn = '/kaggle/input/4637427/train_embeds_esm2_t36_3B_UR50D.npy'\nX = np.load(fn).astype(np.float32)\nprint(X.shape)\nprint(X[:2,:3])\n\n# (MG)\n# Переменная prot_ids более нигде не используется.\n\n# prot_ids  = np.load('/kaggle/input/4637427/train_ids_esm2_t36_3B_UR50D.npy')\n# print(prot_ids.shape)\n# print(prot_ids[:15])\n\n############################ load targets and their ids  ######################################\n\nfn = '/kaggle/input/cafa5-features-etc/Y_1499.npy'\nY = np.load(fn)\nprint(Y.shape)\nprint(Y[:20])\n\n# (MG)\n# Перенес блок загрузки Y_labels в раздел \"Prepare submission\"\n# так как это загрузка меток для финальной проверки, а данные для финальное проверки\n# грузятся в блоке \"Prepare submission\" и получается логический разрыв.\n\n# fn = '/kaggle/input/cafa5-features-etc/Y_1499_labels.npy'\n# print(fn)\n# Y_labels = np.load(fn)\n# print(Y_labels.shape)\n# print(Y_labels[:20])\n\n# (MG)\n# Оч интересные график с сомнительной практической пользой.\n\n# %%time\n\n# if 1:\n#     v = Y.sum(axis = 0)\n#     plt.figure(figsize = (20,6))\n#     plt.plot(v[:500], '*-')\n#     plt.grid()\n#     plt.title(' Number of 1 in targets',fontsize = 20 )\n#     plt.xlabel('target index', fontsize = 20 )\n#     plt.show()\n\n# fn_train_terms = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv'\n# fn_train_taxonomy ='/kaggle/input/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv'","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:03.130525Z","iopub.execute_input":"2023-08-02T15:43:03.131101Z","iopub.status.idle":"2023-08-02T15:43:47.914313Z","shell.execute_reply.started":"2023-08-02T15:43:03.131072Z","shell.execute_reply":"2023-08-02T15:43:47.913044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train test split ","metadata":{}},{"cell_type":"code","source":"%%time\nfrom sklearn.model_selection import train_test_split\n\nIX_train,IX_val = train_test_split(np.arange(len(X)), train_size=0.7, random_state=42)\nprint(IX_train.shape,IX_val.shape)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:47.916053Z","iopub.execute_input":"2023-08-02T15:43:47.91655Z","iopub.status.idle":"2023-08-02T15:43:47.930914Z","shell.execute_reply.started":"2023-08-02T15:43:47.916512Z","shell.execute_reply":"2023-08-02T15:43:47.929783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create X/Y_train, X/Y_val as pytorch tensors and send to device","metadata":{}},{"cell_type":"code","source":"%%time\nX_train = torch.tensor(X[IX_train,:], dtype=torch.float32, device=device)\nY_train = torch.tensor(Y[IX_train,:], dtype=torch.float32, device=device)\n\nX_val = torch.tensor(X[IX_val,:], dtype=torch.float32, device=device)\nY_val = torch.tensor(Y[IX_val,:], dtype=torch.float32, device=device)\n\nprint(X_train.shape, Y_train.shape, X_val.shape, Y_val.shape )","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:47.932226Z","iopub.execute_input":"2023-08-02T15:43:47.93254Z","iopub.status.idle":"2023-08-02T15:43:50.208584Z","shell.execute_reply.started":"2023-08-02T15:43:47.932514Z","shell.execute_reply":"2023-08-02T15:43:50.207515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Wrap X,Y into \"TensorDataset\" and create \"DataLoader\"\n\n\"TensorDataset\" allows to unify several arrays to work with them as a whole. For example for pair X_train and Y_train we can create  \"train_dataset\"  which combines both. \n\nAs a next step we create \"DataLoader\" object from \"train_dataset\" which automatically take care about batches - see the main training loop of the neural network","metadata":{}},{"cell_type":"code","source":"# BATCH_SIZE = 5120\n# EPOCHS = 50\n# LEARNING_RATE = 0.001\n# MOMENTUM = 0.9\n# OPT_FUNC = torch.optim.Adam","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:50.210057Z","iopub.execute_input":"2023-08-02T15:43:50.211061Z","iopub.status.idle":"2023-08-02T15:43:50.216297Z","shell.execute_reply.started":"2023-08-02T15:43:50.211028Z","shell.execute_reply":"2023-08-02T15:43:50.214838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nBATCH_SIZE = 64\n\ntrain_dataset = torch.utils.data.TensorDataset(X_train, Y_train)\ntrain_dataloader = torch.utils.data.DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True)\n\n# val_dataset = torch.utils.data.TensorDataset(X_val, Y_val)\n# val_dataloader = torch.utils.data.DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:50.217856Z","iopub.execute_input":"2023-08-02T15:43:50.218196Z","iopub.status.idle":"2023-08-02T15:43:50.231678Z","shell.execute_reply.started":"2023-08-02T15:43:50.218157Z","shell.execute_reply":"2023-08-02T15:43:50.230471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define neural network ","metadata":{}},{"cell_type":"code","source":"# (MG) Перенес на верх.\n\n# import torch.nn as nn","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:50.233073Z","iopub.execute_input":"2023-08-02T15:43:50.233354Z","iopub.status.idle":"2023-08-02T15:43:50.243508Z","shell.execute_reply.started":"2023-08-02T15:43:50.23333Z","shell.execute_reply":"2023-08-02T15:43:50.242513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/alexandervc/moa-nn-04/notebook\n# class MyModel(nn.Module):\n\n#     def __init__(self, in_features, out_features, neurons_per_layer = 2000):\n\n#         super().__init__()\n\n#         self.in_features = in_features\n#         self.out_features = out_features\n\n#         self.model = nn.Sequential(\n#             nn.Dropout(0.2),\n#             nn.BatchNorm1d(in_features),\n#             nn.Linear(in_features, neurons_per_layer),\n#             nn.LayerNorm(neurons_per_layer),\n#             nn.PReLU(init = 0.5),\n\n#             nn.Dropout(0.2),\n#             nn.BatchNorm1d(neurons_per_layer),\n#             nn.Linear(neurons_per_layer, neurons_per_layer),\n#             nn.LayerNorm(neurons_per_layer),\n#             nn.PReLU(init = 0.5),\n\n#             nn.BatchNorm1d(neurons_per_layer),\n#             nn.Linear(neurons_per_layer, out_features),\n#             nn.LayerNorm(out_features),\n#             nn.Sigmoid()\n#         )\n\n#     def forward(self, x):\n#         return self.model(x)\n\nclass MyModel(nn.Module):\n\n    def __init__(self, in_features, out_features, neurons_per_layer = 2000):\n\n        super().__init__()\n\n        self.in_features = in_features\n        self.out_features = out_features\n\n        # in_features -> neurons_per_layer\n\n        self.step1 = nn.Sequential(\n            nn.Dropout(0.2, inplace = True),\n            nn.BatchNorm1d(in_features),\n            nn.Linear(in_features, neurons_per_layer),\n            nn.LayerNorm(neurons_per_layer),\n            nn.PReLU(init = 0.5),\n        )\n\n        # neurons_per_layer -> neurons_per_layer\n\n        self.step2 = nn.Sequential(\n            nn.Dropout(0.2, inplace = True),\n            nn.BatchNorm1d(neurons_per_layer),\n            nn.Linear(neurons_per_layer, neurons_per_layer),\n            nn.LayerNorm(neurons_per_layer),\n            nn.PReLU(init = 0.5),\n        )\n\n        # neurons_per_layer * 2 -> out_features\n\n        self.step3 = nn.Sequential(\n            nn.BatchNorm1d(neurons_per_layer * 2),\n            nn.Linear(neurons_per_layer * 2, out_features),\n            nn.LayerNorm(out_features),\n            nn.Sigmoid()\n        )\n\n    def forward(self, x):\n        step1 = self.step1(x)\n        step2 = self.step2(step1)\n        x = torch.cat([step2, step1], axis = -1)\n        return self.step3(x)\n\nmodel = MyModel(X_train.shape[1], Y_train.shape[1]).to(device)  ","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:50.245453Z","iopub.execute_input":"2023-08-02T15:43:50.245889Z","iopub.status.idle":"2023-08-02T15:43:50.453986Z","shell.execute_reply.started":"2023-08-02T15:43:50.245852Z","shell.execute_reply":"2023-08-02T15:43:50.452351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model to device","metadata":{}},{"cell_type":"code","source":"# (MG) \n# Модель созадется для непонятного колдунства которое будет следом.\n# Далее в блоке обучения модель создается снова, и так как не вижу смысла создовать модель два раза\n# закрываю это создание комментариями.\n\n# model = MyModel(X.shape[1],Y.shape[1])\n# model.to(device)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:50.455379Z","iopub.execute_input":"2023-08-02T15:43:50.455966Z","iopub.status.idle":"2023-08-02T15:43:50.460587Z","shell.execute_reply.started":"2023-08-02T15:43:50.455935Z","shell.execute_reply":"2023-08-02T15:43:50.459571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# (MG)\n# Видимо тут проверки на невероятные параметры \n# среди которых переменная XX которая не была объявлена ранее. \n\n# (MG)\n# Как оказалось torchsummary это функция отображения модели красиво. \n\n\n# That does not work for us - why ? \n\n# try:\n#     torchsummary(model, XX.size(), batch_size=-1, device='cuda')\n# except    Exception as inst:\n#     print(type(inst))    # the exception type\n#     print(inst.args)     # arguments stored in .args\n#     print(inst)          # __str__ allows args to be printed directly,\n#                          # but may be overridden in exception subclasses\n","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:50.462276Z","iopub.execute_input":"2023-08-02T15:43:50.462601Z","iopub.status.idle":"2023-08-02T15:43:50.472242Z","shell.execute_reply.started":"2023-08-02T15:43:50.462575Z","shell.execute_reply":"2023-08-02T15:43:50.471468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train neural network ","metadata":{}},{"cell_type":"code","source":"import datetime\ncurrent_datetime = datetime.datetime.now()\n\nprint(\"Current datetime:\", current_datetime)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:50.473303Z","iopub.execute_input":"2023-08-02T15:43:50.474017Z","iopub.status.idle":"2023-08-02T15:43:50.489341Z","shell.execute_reply.started":"2023-08-02T15:43:50.473986Z","shell.execute_reply":"2023-08-02T15:43:50.48821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nmax_epoch = 10\n\nimport time \nfrom torchmetrics.classification import BinaryF1Score\nfrom torchmetrics import AUROC\n\nt0 = time.time()\n\nf1 = BinaryF1Score(threshold=0.25, multidim_average= 'samplewise'  )\nauroc = AUROC(task = 'binary')\n\nprint(f\"X.shape {X.shape} , Y.shape {Y.shape}\")\nprint()\n\n# do not fortget to reinitialize model when you debug the notebook, since otherwise - each time you rerun the current cell \n# - you will update existing weights  - it would be easy to see that starting loss is unexpectdely small \nmodel = MyModel(X.shape[1],Y.shape[1])\nmodel.to(device)\n\n# Инициализация весов\n\n# def weights_init(m):\n#     classname = m.__class__.__name__\n#     if classname.find(\"Linear\") != -1:\n#         torch.nn.init.normal_(m.weight.data, 0.0, 0.03)\n\n#     # if classname.find(\"BatchNorm\") != -1:\n#     #     torch.nn.init.normal_(m.weight.data, 1.0, 0.02)\n#     #     torch.nn.init.constant_(m.bias.data, 0.0)\n\n# model.apply(weights_init)\n\noptimizer = torch.optim.Adam(\n    model.parameters(),\n    lr=0.005\n) #SGD(model.parameters(), lr=1e-3)\n\n# scheduler = torch.optim.lr_scheduler.StepLR(optimizer, 0.05, 10)\n\ncriterion = nn.BCELoss()\n\ndf_stat = pd.DataFrame(); IX_stat = -1;\nprint()\ncurrent_datetime = datetime.datetime.now()\nprint('Start training NN', str(model))\nprint( current_datetime)\nprint()\n\nfor epoch in range(max_epoch):\n    \n    t0_epoch = time.time()\n    \n    model.train()\n    \n    for i, (x_batch, y_batch) in enumerate(train_dataloader):\n        x_batch, y_batch = x_batch.to(device), y_batch.to(device)\n        optimizer.zero_grad()\n        preds = model(x_batch)\n        loss = criterion(preds, y_batch)\n        loss.backward()\n        optimizer.step()\n        if i % 100 == 0:\n            print(f'Epoch: {epoch}, batch: {i},  train loss on batch: {loss.item():12.5f} , time: {time.time() - t0:.1f} ' )\n\n    t0_epoch_train = time.time() - t0_epoch\n    \n    model.eval()\n    \n    with torch.no_grad():\n        train, y_train = train_dataset.tensors\n        # train, y_train = train.to(device), y_train.to(device) # They already should on device, why we need that ?\n        preds = model(train)\n        preds_train = preds\n        loss = criterion(preds, y_train)\n        loss1 = loss.item()\n        print()\n        print(f'Epoch: {epoch} finished,  train loss: {loss.item():12.5f} , time: {time.time() - t0:.1f} seconds ' )\n        \n        preds = model(X_val)\n        preds_val = preds\n        loss = criterion(preds, Y_val)\n        loss2 = loss.item()\n        print(f'Epoch: {epoch} finished,  VALidation loss: {loss.item():12.5f} , time: {time.time() - t0:.1f} seconds ' )\n        print()\n        \n        IX_stat += 1\n        df_stat.loc[IX_stat, 'epoch'] = epoch\n        df_stat.loc[IX_stat, 'BCE Val'] = loss2\n        df_stat.loc[IX_stat, 'BCE Train'] = loss1\n        \n        #from torchmetrics.classification import BinaryF1Score\n        f1 = BinaryF1Score(threshold=0.25, multidim_average= 'samplewise'  )\n        s = f1(preds_val, Y_val) # vector of values for each sample\n        df_stat.loc[IX_stat, 'F1|0.25 Val'] = s.mean().item()#  average over samples\n        s = f1(preds_train, y_train) \n        df_stat.loc[IX_stat, 'F1|0.25 Train'] =  s.mean().item()\n        \n        #from torchmetrics import AUROC\n        auroc = AUROC(task = 'binary')\n        s = auroc(preds_val, Y_val) # vector of values for each sample\n        df_stat.loc[IX_stat, 'AUC Val'] = s.item()#  average over samples\n        s = auroc(preds_train, y_train) \n        df_stat.loc[IX_stat, 'AUC Train'] =  s.item()\n        \n        \n        df_stat.loc[IX_stat, 'Time epoch train'] = np.round(t0_epoch_train , 1 )\n        df_stat.loc[IX_stat, 'Time epoch full'] = np.round(time.time() - t0_epoch , 1 )\n        display(df_stat.tail(2))\n        \n\n        \nprint('Training finished', '%.1f seconds passed'%(time.time() - t0 ))\ndf_stat = df_stat.round(4)\ndisplay(df_stat)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T15:43:50.491804Z","iopub.execute_input":"2023-08-02T15:43:50.492138Z","iopub.status.idle":"2023-08-02T16:34:47.923429Z","shell.execute_reply.started":"2023-08-02T15:43:50.49211Z","shell.execute_reply":"2023-08-02T16:34:47.921234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_metrics_keywords = ['BCE','F1', 'AUC']\n\nplt.figure(figsize = (20,8) )\nfor i,kw in enumerate(list_metrics_keywords):\n    l = [col for col in df_stat if kw in col]\n    plt.subplot(1,len(list_metrics_keywords ),i+1)\n    plt.plot( df_stat[l], label = l )\n    plt.grid()\n    plt.legend(fontsize = 12)\n    plt.title(kw, fontsize = 20)\n    plt.xlabel('epoch',fontsize = 15)\n    \nplt.show() \n","metadata":{"execution":{"iopub.status.busy":"2023-08-02T16:34:47.927544Z","iopub.execute_input":"2023-08-02T16:34:47.927999Z","iopub.status.idle":"2023-08-02T16:34:48.859511Z","shell.execute_reply.started":"2023-08-02T16:34:47.927967Z","shell.execute_reply":"2023-08-02T16:34:48.858707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Comments\n\nThe actual metric in the CAFA5 competition is similar to F1 above, but more sophisticated and computation time is signinificantly large. One may use F1 with threshold around 0.2-0.3 as a kind of \"fast-proxi\" for it.\n\nThe next step to get close to actual CAFA5 metric is to calculate F1 with different thresholds for different subontologies and choose not fixed threhold, but optimal one. Still it would not be exactly CAFA5 F1. Extra steps are: weighting and propagation. We will not go into these details here.","metadata":{}},{"cell_type":"markdown","source":"# Clear memory ","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\ndel X_train,Y_train, X_val, Y_val, train_dataset, train_dataloader, X, Y\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2023-08-02T16:34:48.861047Z","iopub.execute_input":"2023-08-02T16:34:48.861358Z","iopub.status.idle":"2023-08-02T16:34:49.976337Z","shell.execute_reply.started":"2023-08-02T16:34:48.861331Z","shell.execute_reply":"2023-08-02T16:34:49.975198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare submission","metadata":{}},{"cell_type":"markdown","source":"## Load features for submission","metadata":{"execution":{"iopub.status.busy":"2023-05-30T15:38:23.584425Z","iopub.execute_input":"2023-05-30T15:38:23.584794Z","iopub.status.idle":"2023-05-30T15:59:43.181211Z","shell.execute_reply.started":"2023-05-30T15:38:23.584742Z","shell.execute_reply":"2023-05-30T15:59:43.179867Z"}}},{"cell_type":"code","source":"%%time \n################ load  features #########################\n\n# fn = '/kaggle/input/4637427/train_embeds_esm2_t36_3B_UR50D.npy'\nfn = '/kaggle/input/4637427/test_embeds_esm2_t36_3B_UR50D.npy'\nprint(fn)\nX = np.load(fn).astype(np.float32)\nprint(X.shape)\nprint(X[:2,:3])\n\nfn = '/kaggle/input/cafa5-features-etc/Y_1499_labels.npy'\nprint(fn)\nY_labels = np.load(fn)\nprint(Y_labels.shape)\nprint(Y_labels[:20])","metadata":{"execution":{"iopub.status.busy":"2023-08-02T16:34:49.978285Z","iopub.execute_input":"2023-08-02T16:34:49.978718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nX = torch.tensor(X, dtype=torch.float32).to(device)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Compute predictions","metadata":{}},{"cell_type":"code","source":"%%time\nmodel.eval()\nwith torch.no_grad():\n    preds = model(X)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(preds.shape)\nprint(preds[:4,:3])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest_protein_ids = np.load('/kaggle/input/4637427/test_ids_esm2_t36_3B_UR50D.npy')\nprint(test_protein_ids.shape, test_protein_ids[:10])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare submission tsv file","metadata":{}},{"cell_type":"code","source":"%%time\ndf_submission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])\n\nn_targets_predicted = preds.shape[1]\nn_samples_predicted = preds.shape[0]\nprint('n_samples_predicted, n_targets_predicted',  n_samples_predicted, n_targets_predicted )\n\n\nprotein_list = []\nfor k in list(test_protein_ids):\n    protein_list += [k] * n_targets_predicted\ndf_submission['Protein Id'] = protein_list\n\ndf_submission['GO Term Id'] = list(Y_labels) * n_samples_predicted\ndf_submission['Prediction'] = preds.ravel()\n\ndf_submission = df_submission.round(3)\ndf_submission = df_submission[ df_submission['Prediction'] > 0.01  ]\n\nmemory_usage_per_column = df_submission.memory_usage(deep=True)\ntotal_memory_usage = memory_usage_per_column.sum()\nprint(\"\\nTotal memory usage:\", total_memory_usage/1e6, \"Megabytes\")\n\nprint(df_submission.shape)\ndisplay(df_submission)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport gc\nif 0:\n    del preds \n    \ngc.collect()\n\ndf_submission.to_csv(\"submission.tsv\", header=False, index=False, sep='\\t')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nprint(df_submission.shape)\nplt.figure(figsize = (15,4))\nplt.hist(df_submission['Prediction'].values, bins = 1000 )\nplt.show()\nprint(df_submission.shape)\ndisplay(df_submission.describe())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for t in [0.1,0.2, 0.25,0.28, 0.3,0.4,0.5,0.6,0.7,0.8,0.9,1]:\n    m = df_submission['Prediction'] > t\n    print(t, m.sum(), m.sum()/ (n_samples_predicted * n_targets_predicted ) )\n    \nprint()    \ntry:\n    print( Y.sum(),  Y.sum()/ (Y.shape[0] * Y.shape[1]) )\nexcept:\n    pass    \n\nprint('Here is fast rationale why we should think of threshold for F1 is around 0.28 - number of 1 in that case corresponds to train data')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"print('%.1f seconds passed total '%(time.time()-t0start) )","metadata":{}}]}