{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":41875,"databundleVersionId":5521661,"sourceType":"competition"},{"sourceId":5549164,"sourceType":"datasetVersion","datasetId":3197305},{"sourceId":5792099,"sourceType":"datasetVersion","datasetId":3327296},{"sourceId":6247561,"sourceType":"datasetVersion","datasetId":3590060}],"dockerImageVersionId":30527,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nsub = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/sample_submission.tsv\", sep= \"\\t\", header = None)\nsub.columns = [\"The Protein ID\", \"The Gene Ontology term (GO) ID\", \"Predicted link probability that GO appear in Protein\"]\nsub.head(5)","metadata":{"papermill":{"duration":0.319043,"end_time":"2023-07-23T05:42:41.573546","exception":false,"start_time":"2023-07-23T05:42:41.254503","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:05:39.532599Z","iopub.execute_input":"2023-08-04T11:05:39.532969Z","iopub.status.idle":"2023-08-04T11:05:39.765631Z","shell.execute_reply.started":"2023-08-04T11:05:39.532937Z","shell.execute_reply":"2023-08-04T11:05:39.764646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAIN_DIR = \"/kaggle/input/cafa-5-protein-function-prediction\"\n\nimport numpy as np\nfrom tqdm import tqdm\nimport time\nimport matplotlib.pyplot as plt\nplt.style.use('ggplot')\nimport torch\nfrom torch.utils.data import Dataset\nfrom torch import nn\nfrom torch.utils.data import random_split\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nfrom torchmetrics.classification import MultilabelF1Score\nfrom torchmetrics.classification import MultilabelAccuracy\nimport pytorch_lightning as pl\nfrom pytorch_lightning import Trainer\nfrom pytorch_lightning.loggers import WandbLogger\nimport wandb\nimport os\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":16.111487,"end_time":"2023-07-23T05:42:57.692877","exception":false,"start_time":"2023-07-23T05:42:41.58139","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:05:39.76725Z","iopub.execute_input":"2023-08-04T11:05:39.767563Z","iopub.status.idle":"2023-08-04T11:05:58.205481Z","shell.execute_reply.started":"2023-08-04T11:05:39.767538Z","shell.execute_reply":"2023-08-04T11:05:58.204353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    train_sequences_path = MAIN_DIR  + \"/Train/train_sequences.fasta\"\n    train_labels_path = MAIN_DIR + \"/Train/train_terms.tsv\"\n    test_sequences_path = MAIN_DIR + \"/Test (Targets)/testsuperset.fasta\"\n    \n    num_labels = 500\n    n_epochs = 20\n    batch_size = 128\n    lr = 0.1\n    \n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')","metadata":{"papermill":{"duration":0.01773,"end_time":"2023-07-23T05:42:57.719021","exception":false,"start_time":"2023-07-23T05:42:57.701291","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:05:58.207241Z","iopub.execute_input":"2023-08-04T11:05:58.207686Z","iopub.status.idle":"2023-08-04T11:05:58.213829Z","shell.execute_reply.started":"2023-08-04T11:05:58.207651Z","shell.execute_reply":"2023-08-04T11:05:58.212875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(config.device)","metadata":{"papermill":{"duration":0.018179,"end_time":"2023-07-23T05:42:57.745526","exception":false,"start_time":"2023-07-23T05:42:57.727347","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:05:58.216163Z","iopub.execute_input":"2023-08-04T11:05:58.216758Z","iopub.status.idle":"2023-08-04T11:05:58.235198Z","shell.execute_reply.started":"2023-08-04T11:05:58.216724Z","shell.execute_reply":"2023-08-04T11:05:58.233904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeds_map = {\n    \"T5\" : \"t5embeds\",\n    \"ProtBERT\" : \"protbert-embeddings-for-cafa5\",\n    \"EMS2\" : \"cafa-5-ems-2-embeddings-numpy\"\n}\n\nembeds_dim = {\n    \"T5\" : 1024,\n    \"ProtBERT\" : 1024,\n    \"EMS2\" : 1280\n}","metadata":{"papermill":{"duration":0.019202,"end_time":"2023-07-23T05:42:57.772957","exception":false,"start_time":"2023-07-23T05:42:57.753755","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:07:09.665363Z","iopub.execute_input":"2023-08-04T11:07:09.666158Z","iopub.status.idle":"2023-08-04T11:07:09.675911Z","shell.execute_reply.started":"2023-08-04T11:07:09.666078Z","shell.execute_reply":"2023-08-04T11:07:09.673833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ProteinSequenceDataset(Dataset):\n    \n    def __init__(self, datatype, embeddings_source):\n        super(ProteinSequenceDataset).__init__()\n        self.datatype = datatype\n        \n        if embeddings_source in [\"ProtBERT\", \"EMS2\"]:\n            embeds = np.load(\"/kaggle/input/\" + embeds_map[embeddings_source] + \"/\" + datatype + \"_embeddings.npy\")\n            ids = np.load(\"/kaggle/input/\" + embeds_map[embeddings_source] + \"/\" + datatype + \"_ids.npy\")\n        \n        if embeddings_source == \"T5\":\n            embeds = np.load(\"/kaggle/input/\" + embeds_map[embeddings_source] + \"/\" + datatype + \"_embeds.npy\")\n            ids = np.load(\"/kaggle/input/\" + embeds_map[embeddings_source] + \"/\" + datatype + \"_ids.npy\")\n            \n        embeds_list = []\n        for l in range(embeds.shape[0]):\n            embeds_list.append(embeds[l,:])\n        self.df = pd.DataFrame(data={\"EntryID\": ids, \"embed\" : embeds_list})\n        \n        if datatype == \"train\":\n            np_labels = np.load(\n                \"/kaggle/input/train-targets-top\" + str(config.num_labels) + \\\n                \"/train_targets_top\" + str(config.num_labels) + \".npy\")\n            df_labels = pd.DataFrame(self.df['EntryID'])\n            df_labels['labels_vect'] = [row for row in np_labels]\n            self.df = self.df.merge(df_labels, on = \"EntryID\")\n            \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        embed = torch.tensor(self.df.iloc[index][\"embed\"] , dtype = torch.float32)\n        if self.datatype==\"train\":\n            targets = torch.tensor(self.df.iloc[index][\"labels_vect\"], dtype = torch.float32)\n            return embed, targets\n        if self.datatype == \"test\":\n            id = self.df.iloc[index][\"EntryID\"]\n            return embed, id\n        ","metadata":{"papermill":{"duration":0.025424,"end_time":"2023-07-23T05:42:57.806509","exception":false,"start_time":"2023-07-23T05:42:57.781085","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:05:58.254232Z","iopub.execute_input":"2023-08-04T11:05:58.254826Z","iopub.status.idle":"2023-08-04T11:05:58.271296Z","shell.execute_reply.started":"2023-08-04T11:05:58.254793Z","shell.execute_reply":"2023-08-04T11:05:58.270325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## MODEL ARCHITECTURE BUILDING AND TRAINING","metadata":{"papermill":{"duration":0.007662,"end_time":"2023-07-23T05:42:57.822448","exception":false,"start_time":"2023-07-23T05:42:57.814786","status":"completed"},"tags":[]}},{"cell_type":"code","source":"class MultiLayerPerceptron(torch.nn.Module):\n\n    def __init__(self, input_dim, num_classes):\n        super(MultiLayerPerceptron, self).__init__()\n\n        self.linear1 = torch.nn.Linear(input_dim, 864)\n        self.activation1 = torch.nn.ReLU() \n        self.linear2 = torch.nn.Linear(864, 712)\n        self.activation2 = torch.nn.ReLU()\n        self.linear3 = torch.nn.Linear(712, num_classes)\n      \n\n    def forward(self, x):\n        x = self.linear1(x)\n        x = self.activation1(x)\n        x = self.linear2(x)\n        x = self.activation2(x)\n        x = self.linear3(x)\n        return x","metadata":{"papermill":{"duration":0.019945,"end_time":"2023-07-23T05:42:57.850376","exception":false,"start_time":"2023-07-23T05:42:57.830431","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:05:58.272764Z","iopub.execute_input":"2023-08-04T11:05:58.273077Z","iopub.status.idle":"2023-08-04T11:05:58.292723Z","shell.execute_reply.started":"2023-08-04T11:05:58.273051Z","shell.execute_reply":"2023-08-04T11:05:58.291786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CNN1D(nn.Module):\n\n    def __init__(self, input_dim, num_classes):\n        super(CNN1D, self).__init__()\n        self.conv1 = nn.Conv1d(in_channels = 1, out_channels = 3, kernel_size = 5, dilation = 1, padding = 2, stride = 1)\n        self.pool1 = nn.MaxPool1d(kernel_size = 2, stride = 2)\n        self.conv2 = nn.Conv1d(in_channels = 3, out_channels = 8, kernel_size = 5, dilation = 1, padding = 2, stride = 1)\n        self.pool2 = nn.MaxPool1d(kernel_size = 2, stride = 2)\n        self.fc1 = nn.Linear(in_features = int(8 * input_dim / 4), out_features = 864)\n        self.fc2 = nn.Linear(in_features = 864, out_features = num_classes)\n\n\n    def forward(self, x):\n        x = x.reshape(x.shape[0], 1, x.shape[1])\n        x = self.pool1(nn.functional.tanh(self.conv1(x)))\n        x = self.pool2(nn.functional.tanh(self.conv2(x)))\n        x = torch.flatten(x, 1)\n        x = nn.functional.tanh(self.fc1(x))\n        x = self.fc2(x)\n        return x\n","metadata":{"papermill":{"duration":0.023288,"end_time":"2023-07-23T05:42:57.881913","exception":false,"start_time":"2023-07-23T05:42:57.858625","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:05:58.294005Z","iopub.execute_input":"2023-08-04T11:05:58.294418Z","iopub.status.idle":"2023-08-04T11:05:58.316608Z","shell.execute_reply.started":"2023-08-04T11:05:58.294335Z","shell.execute_reply":"2023-08-04T11:05:58.315611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_model(embeddings_source, model_type = \"linear\", train_size = 0.7):\n    \n    train_dataset = ProteinSequenceDataset(datatype = \"train\", embeddings_source = embeddings_source)\n    \n    train_set, val_set = random_split(train_dataset, lengths = [int(len(train_dataset) * train_size), len(train_dataset) - int(len(train_dataset) * train_size)])\n    train_dataloader = torch.utils.data.DataLoader(train_set, batch_size = config.batch_size, shuffle = True)\n    val_dataloader = torch.utils.data.DataLoader(val_set, batch_size = config.batch_size, shuffle = True)\n\n    if model_type == \"linear\":\n        model = MultiLayerPerceptron(input_dim = embeds_dim[embeddings_source], num_classes = config.num_labels).to(config.device)\n    if model_type == \"convolutional\":\n        model = CNN1D(input_dim = embeds_dim[embeddings_source], num_classes = config.num_labels).to(config.device)\n\n    optimizer = torch.optim.Adam(model.parameters(), lr = config.lr)\n    scheduler = ReduceLROnPlateau(optimizer, factor = 0.1, patience = 1)\n    CrossEntropy = torch.nn.CrossEntropyLoss()\n    f1_score = MultilabelF1Score(num_labels = config.num_labels).to(config.device)\n    n_epochs = config.n_epochs\n\n    print(\"BEGIN TRAINING...\")\n    train_loss_history = []\n    val_loss_history = []\n    \n    train_f1score_history = []\n    val_f1score_history = []\n    for epoch in range(n_epochs):\n        print(\"EPOCH \", epoch + 1)\n        losses = []\n        scores = []\n        for embed, targets in tqdm(train_dataloader):\n            embed, targets = embed.to(config.device), targets.to(config.device)\n            optimizer.zero_grad()\n            preds = model(embed)\n            loss = CrossEntropy(preds, targets)\n            score = f1_score(preds, targets)\n            losses.append(loss.item()) \n            scores.append(score.item())\n            loss.backward()\n            optimizer.step()\n        avg_loss = np.mean(losses)\n        avg_score = np.mean(scores)\n        print(\"Running Average TRAIN Loss : \", avg_loss)\n        print(\"Running Average TRAIN F1-Score : \", avg_score)\n        train_loss_history.append(avg_loss)\n        train_f1score_history.append(avg_score)\n        \n        losses = []\n        scores = []\n        for embed, targets in val_dataloader:\n            embed, targets = embed.to(config.device), targets.to(config.device)\n            preds = model(embed)\n            loss = CrossEntropy(preds, targets)\n            score = f1_score(preds, targets)\n            losses.append(loss.item())\n            scores.append(score.item())\n        avg_loss = np.mean(losses)\n        avg_score = np.mean(scores)\n        print(\"Running Average VAL Loss : \", avg_loss)\n        print(\"Running Average VAL F1-Score : \", avg_score)\n        val_loss_history.append(avg_loss)\n        val_f1score_history.append(avg_score)\n        \n        scheduler.step(avg_loss)\n        print(\"\\n\")\n        \n    print(\"TRAINING FINISHED\")\n    print(\"FINAL TRAINING SCORE : \", train_f1score_history[-1])\n    print(\"FINAL VALIDATION SCORE : \", val_f1score_history[-1])\n    \n    losses_history = {\"train\" : train_loss_history, \"val\" : val_loss_history}\n    scores_history = {\"train\" : train_f1score_history, \"val\" : val_f1score_history}\n    \n    return model, losses_history, scores_history","metadata":{"papermill":{"duration":0.03402,"end_time":"2023-07-23T05:42:57.924143","exception":false,"start_time":"2023-07-23T05:42:57.890123","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:05:58.318182Z","iopub.execute_input":"2023-08-04T11:05:58.318797Z","iopub.status.idle":"2023-08-04T11:05:58.337339Z","shell.execute_reply.started":"2023-08-04T11:05:58.318764Z","shell.execute_reply":"2023-08-04T11:05:58.335672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ems2_model, ems2_losses, ems2_scores = train_model(embeddings_source = \"EMS2\",model_type = \"linear\")","metadata":{"papermill":{"duration":0.017195,"end_time":"2023-07-23T05:42:57.949671","exception":false,"start_time":"2023-07-23T05:42:57.932476","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:07:14.862491Z","iopub.execute_input":"2023-08-04T11:07:14.862855Z","iopub.status.idle":"2023-08-04T11:12:50.86262Z","shell.execute_reply.started":"2023-08-04T11:07:14.862829Z","shell.execute_reply":"2023-08-04T11:12:50.861127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#t5_model, t5_losses, t5_scores = train_model(embeddings_source = \"T5\", model_type = \"convolutional\")","metadata":{"papermill":{"duration":0.016067,"end_time":"2023-07-23T05:42:57.974108","exception":false,"start_time":"2023-07-23T05:42:57.958041","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:05:59.199958Z","iopub.status.idle":"2023-08-04T11:05:59.200429Z","shell.execute_reply.started":"2023-08-04T11:05:59.200278Z","shell.execute_reply":"2023-08-04T11:05:59.200295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#protbert_model, protbert_losses, protbert_scores = train_model(embeddings_source=\"ProtBERT\",model_type=\"linear\")","metadata":{"papermill":{"duration":341.48694,"end_time":"2023-07-23T05:48:39.469256","exception":false,"start_time":"2023-07-23T05:42:57.982316","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:05:59.201127Z","iopub.status.idle":"2023-08-04T11:05:59.202392Z","shell.execute_reply.started":"2023-08-04T11:05:59.202005Z","shell.execute_reply":"2023-08-04T11:05:59.202048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10, 4))\nplt.plot(ems2_losses[\"val\"], label = \"EMS2\")\n#plt.plot(t5_losses[\"val\"], label = \"T5\")\n#plt.plot(ems2_losses[\"val\"], label = \"EMS2\") \nplt.title(\"Validation Losses for # Vector Embeddings\")\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Average Loss\")\nplt.legend()\nplt.show()\n\nplt.figure(figsize = (10, 4))\n#plt.plot(ems2_scores[\"val\"], label = \"EMS2\")\n#plt.plot(t5_scores[\"val\"], label = \"T5\")\nplt.plot(ems2_scores[\"val\"], label = \"EMS2\")\nplt.title(\"Validation F1-Scores for # Vector Embeddings\")\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Average F1-Score\")\nplt.legend()\nplt.show()","metadata":{"papermill":{"duration":0.921151,"end_time":"2023-07-23T05:48:40.624003","exception":false,"start_time":"2023-07-23T05:48:39.702852","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:12:50.866015Z","iopub.execute_input":"2023-08-04T11:12:50.86652Z","iopub.status.idle":"2023-08-04T11:12:51.371193Z","shell.execute_reply.started":"2023-08-04T11:12:50.866486Z","shell.execute_reply":"2023-08-04T11:12:51.369691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## PREDICTION","metadata":{"papermill":{"duration":0.222472,"end_time":"2023-07-23T05:48:41.132846","exception":false,"start_time":"2023-07-23T05:48:40.910374","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def predict(embeddings_source):\n    \n    test_dataset = ProteinSequenceDataset(datatype=\"test\", embeddings_source = embeddings_source)\n    test_dataloader = torch.utils.data.DataLoader(test_dataset, batch_size = 1, shuffle = False)\n    \n    if embeddings_source == \"T5\":\n        model = t5_model\n    if embeddings_source == \"ProtBERT\":\n        model = protbert_model\n    if embeddings_source == \"EMS2\":\n        model = ems2_model\n        \n    model.eval()\n    \n    labels = pd.read_csv(config.train_labels_path, sep = \"\\t\")\n    top_terms = labels.groupby(\"term\")[\"EntryID\"].count().sort_values(ascending=False)\n    labels_names = top_terms[:config.num_labels].index.values\n    print(\"GENERATE PREDICTION FOR TEST SET...\")\n\n    ids_ = np.empty(shape=(len(test_dataloader)*config.num_labels,), dtype=object)\n    go_terms_ = np.empty(shape=(len(test_dataloader)*config.num_labels,), dtype=object)\n    confs_ = np.empty(shape=(len(test_dataloader)*config.num_labels,), dtype=np.float32)\n\n    for i, (embed, id) in tqdm(enumerate(test_dataloader)):\n        embed = embed.to(config.device)\n        confs_[i*config.num_labels:(i+1)*config.num_labels] = torch.nn.functional.sigmoid(model(embed)).squeeze().detach().cpu().numpy()\n        ids_[i*config.num_labels:(i+1)*config.num_labels] = id[0]\n        go_terms_[i*config.num_labels:(i+1)*config.num_labels] = labels_names\n\n    submission_df = pd.DataFrame(data={\"Id\" : ids_, \"GO term\" : go_terms_, \"Confidence\" : confs_})\n    print(\"PREDICTIONS DONE\")\n    return submission_df","metadata":{"papermill":{"duration":0.239101,"end_time":"2023-07-23T05:48:41.597135","exception":false,"start_time":"2023-07-23T05:48:41.358034","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:12:51.372733Z","iopub.execute_input":"2023-08-04T11:12:51.373052Z","iopub.status.idle":"2023-08-04T11:12:51.38335Z","shell.execute_reply.started":"2023-08-04T11:12:51.373022Z","shell.execute_reply":"2023-08-04T11:12:51.382202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = predict(\"EMS2\")","metadata":{"papermill":{"duration":115.333183,"end_time":"2023-07-23T05:50:37.152835","exception":false,"start_time":"2023-07-23T05:48:41.819652","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:12:51.385755Z","iopub.execute_input":"2023-08-04T11:12:51.38605Z","iopub.status.idle":"2023-08-04T11:15:30.419713Z","shell.execute_reply.started":"2023-08-04T11:12:51.386025Z","shell.execute_reply":"2023-08-04T11:15:30.418639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(submission_df)","metadata":{"papermill":{"duration":0.330429,"end_time":"2023-07-23T05:50:37.78729","exception":false,"start_time":"2023-07-23T05:50:37.456861","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:15:30.421436Z","iopub.execute_input":"2023-08-04T11:15:30.421797Z","iopub.status.idle":"2023-08-04T11:15:30.430775Z","shell.execute_reply.started":"2023-08-04T11:15:30.421762Z","shell.execute_reply":"2023-08-04T11:15:30.429013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h4> SUBMISSION</h4>\nYou may submit this file as is.","metadata":{"papermill":{"duration":0.365577,"end_time":"2023-07-23T05:50:38.461888","exception":false,"start_time":"2023-07-23T05:50:38.096311","status":"completed"},"tags":[]}},{"cell_type":"code","source":"#submission_df.to_csv('submission.tsv', sep='\\t', header=False, index=False)","metadata":{"papermill":{"duration":0.319918,"end_time":"2023-07-23T05:50:39.088188","exception":false,"start_time":"2023-07-23T05:50:38.76827","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:05:59.209037Z","iopub.status.idle":"2023-08-04T11:05:59.209331Z","shell.execute_reply.started":"2023-08-04T11:05:59.209169Z","shell.execute_reply":"2023-08-04T11:05:59.209181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission2 = pd.read_csv('/kaggle/input/blast-quick-sprof-zero-pred/submission.tsv', sep = '\\t', header = None, names = ['Id2', 'GO term2', 'Confidence2']) ","metadata":{"papermill":{"duration":8.684078,"end_time":"2023-07-23T05:50:48.708279","exception":false,"start_time":"2023-07-23T05:50:40.024201","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:15:30.432083Z","iopub.execute_input":"2023-08-04T11:15:30.432372Z","iopub.status.idle":"2023-08-04T11:15:36.248648Z","shell.execute_reply.started":"2023-08-04T11:15:30.432351Z","shell.execute_reply":"2023-08-04T11:15:36.247129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs = submission2.merge(submission_df, left_on = ['Id2', 'GO term2'], right_on = ['Id', 'GO term'], how='outer')","metadata":{"papermill":{"duration":71.025968,"end_time":"2023-07-23T05:52:00.071742","exception":false,"start_time":"2023-07-23T05:50:49.045774","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:15:36.250095Z","iopub.execute_input":"2023-08-04T11:15:36.251306Z","iopub.status.idle":"2023-08-04T11:16:20.656735Z","shell.execute_reply.started":"2023-08-04T11:15:36.25122Z","shell.execute_reply":"2023-08-04T11:16:20.655177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs.drop(['Id', 'GO term'], axis=1, inplace=True)\nsubs['confidence_combined'] = subs.apply(lambda row: row['Confidence2'] if not np.isnan(row['Confidence2']) else row['Confidence'], axis=1)","metadata":{"papermill":{"duration":1484.522673,"end_time":"2023-07-23T06:16:44.90015","exception":false,"start_time":"2023-07-23T05:52:00.377477","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-08-04T11:16:20.658792Z","iopub.execute_input":"2023-08-04T11:16:20.660189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subs[['Id2', 'GO term2', 'confidence_combined']].to_csv('submission.tsv', sep='\\t', header=False, index=False)","metadata":{"papermill":{"duration":318.942906,"end_time":"2023-07-23T06:22:04.193101","exception":false,"start_time":"2023-07-23T06:16:45.250195","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]}]}