{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The notebook uses code from:<br>\nhttps://www.kaggle.com/code/szabo7zoltan/combineembeddings<br>\n\nalso uses code from:<br>\nhttps://www.kaggle.com/code/simonveitner/simple-mlp<br>\n\n\"K-mer\" code from:<br>\nhttps://www.kaggle.com/code/mohdmuttalib/biological-sequence-modeling-with-k-mer-features<br>\n\nuses dataset \"calculated the occurrence of amino acids for each sample\" from:<br>\nhttps://www.kaggle.com/datasets/bejeweled/cafa5-amino-counts\n\nTOP PUBLIC SUBMISSION:<br>\nhttps://www.kaggle.com/code/siddhvr/cafa-5-t5-embeds-ensemble<br>\n\nhttps://www.kaggle.com/code/samusram/leveraging-foldseek<br>\n","metadata":{}},{"cell_type":"markdown","source":"# Begin of notebook https://www.kaggle.com/code/szabo7zoltan/combineembeddings ","metadata":{}},{"cell_type":"markdown","source":"In a recent Zoom lecture on Cafa-5 there was a suggestion of using multiple embeddings. \nThis is a quick notebook to demonstrate how to concatenate embeddings.\nNote that here we only use a portion of the training data: those proteins that appear in all three aspects (BPO, CCO and MFO).\nThe embeddings are from https://www.kaggle.com/datasets/viktorfairuschin/cafa-5-ems-2-embeddings-numpy\nand https://www.kaggle.com/datasets/sergeifironov/t5embeds\nThe notebook also uses code from \nVictor Fairuschin's notebook: \"https://www.kaggle.com/code/viktorfairuschin/esm-2-embeddings-starter","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nfrom sklearn.linear_model import Ridge\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score\nimport numpy as np\nimport pandas as pd\nfrom catboost import CatBoostRegressor, Pool, CatBoostClassifier, CatBoostRegressor\nfrom keras.layers import LeakyReLU\n\nfrom tqdm import tqdm\ntqdm.pandas()\n\nfrom tensorflow import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout\nfrom tensorflow.keras import layers, callbacks\n# measure roc auc score metric \nfrom tensorflow.keras.metrics import AUC\nimport numpy.random as random","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","tags":[],"execution":{"iopub.status.busy":"2023-07-28T13:42:25.889735Z","iopub.execute_input":"2023-07-28T13:42:25.89033Z","iopub.status.idle":"2023-07-28T13:42:38.877773Z","shell.execute_reply.started":"2023-07-28T13:42:25.890262Z","shell.execute_reply":"2023-07-28T13:42:38.876133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/cafa-5-protein-function-prediction \n#          /home/datalab/nfs/competitionCAFA5/kaggle/cafa-5-ems-2-embeddings-numpy\n#          /home/datalab/nfs/competitionCAFA5/kaggle/t5embeds\n#          /home/datalab/nfs/competitionCAFA5/kaggle/protbert-embeddings-for-cafa5\n#          /home/datalab/nfs/competitionCAFA5/kaggle/cafa5-amino-counts\n# kaggle:  ../input/* \n\n\n!ls ../input/* ","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:42:38.880462Z","iopub.execute_input":"2023-07-28T13:42:38.881283Z","iopub.status.idle":"2023-07-28T13:42:40.055449Z","shell.execute_reply.started":"2023-07-28T13:42:38.881237Z","shell.execute_reply":"2023-07-28T13:42:40.053494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Assigning labels","metadata":{}},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/cafa-5-protein-function-prediction \n#          /home/datalab/nfs/competitionCAFA5/kaggle/t5embeds \n# kaggle:  /kaggle/input/cafa-5-protein-function-prediction\n#          /kaggle/input/t5embeds   \n\nDATA_DIR = '/kaggle/input/cafa-5-protein-function-prediction'\nDATA_DIR2 = '/kaggle/input/t5embeds'","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:42:40.057877Z","iopub.execute_input":"2023-07-28T13:42:40.058486Z","iopub.status.idle":"2023-07-28T13:42:40.066408Z","shell.execute_reply.started":"2023-07-28T13:42:40.058434Z","shell.execute_reply":"2023-07-28T13:42:40.064699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms = pd.read_csv(os.path.join(DATA_DIR, 'Train', 'train_terms.tsv'), sep='\\t')\nterms = train_terms.groupby(['aspect', 'term'])['term'].count().reset_index(name='frequency')\nprint(terms.groupby('aspect')['term'].nunique())","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:42:40.07285Z","iopub.execute_input":"2023-07-28T13:42:40.076497Z","iopub.status.idle":"2023-07-28T13:42:46.648357Z","shell.execute_reply.started":"2023-07-28T13:42:40.076235Z","shell.execute_reply":"2023-07-28T13:42:46.646615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:42:46.650556Z","iopub.execute_input":"2023-07-28T13:42:46.651147Z","iopub.status.idle":"2023-07-28T13:42:46.682217Z","shell.execute_reply.started":"2023-07-28T13:42:46.651063Z","shell.execute_reply":"2023-07-28T13:42:46.680567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CCOProt = set(train_terms[train_terms['aspect']=='CCO']['EntryID'].unique())\nlen(CCOProt)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:42:46.684598Z","iopub.execute_input":"2023-07-28T13:42:46.685155Z","iopub.status.idle":"2023-07-28T13:42:47.933767Z","shell.execute_reply.started":"2023-07-28T13:42:46.685077Z","shell.execute_reply":"2023-07-28T13:42:47.932414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MFOProt = set(train_terms[train_terms['aspect']=='MFO']['EntryID'].unique())\nlen(MFOProt)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:42:47.935172Z","iopub.execute_input":"2023-07-28T13:42:47.936285Z","iopub.status.idle":"2023-07-28T13:42:49.08926Z","shell.execute_reply.started":"2023-07-28T13:42:47.936242Z","shell.execute_reply":"2023-07-28T13:42:49.087866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BPOProt = set(train_terms[train_terms['aspect']=='BPO']['EntryID'].unique())\nlen(BPOProt)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:42:49.091644Z","iopub.execute_input":"2023-07-28T13:42:49.092221Z","iopub.status.idle":"2023-07-28T13:42:50.727514Z","shell.execute_reply.started":"2023-07-28T13:42:49.092161Z","shell.execute_reply":"2023-07-28T13:42:50.725997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AllProt = set(train_terms['EntryID'].unique())\nlen(AllProt)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:42:50.729358Z","iopub.execute_input":"2023-07-28T13:42:50.729812Z","iopub.status.idle":"2023-07-28T13:42:51.374288Z","shell.execute_reply.started":"2023-07-28T13:42:50.729765Z","shell.execute_reply":"2023-07-28T13:42:51.37283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FullProt = []\nfor x in MFOProt:\n    if x in CCOProt and x in BPOProt:\n        FullProt.append(x)\nprint(len(FullProt))\nFullProt = set(FullProt)\nprint(len(FullProt))","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:42:51.382663Z","iopub.execute_input":"2023-07-28T13:42:51.383176Z","iopub.status.idle":"2023-07-28T13:42:51.450685Z","shell.execute_reply.started":"2023-07-28T13:42:51.383086Z","shell.execute_reply":"2023-07-28T13:42:51.448808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FullProt = set.intersection(MFOProt, CCOProt, BPOProt)\nlen(FullProt)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:42:51.452671Z","iopub.execute_input":"2023-07-28T13:42:51.453158Z","iopub.status.idle":"2023-07-28T13:42:51.48505Z","shell.execute_reply.started":"2023-07-28T13:42:51.453083Z","shell.execute_reply":"2023-07-28T13:42:51.483204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"V = {}\nV['BPO'] = 1100\nV['CCO'] = 300\nV['MFO'] = 450","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:42:51.487923Z","iopub.execute_input":"2023-07-28T13:42:51.48838Z","iopub.status.idle":"2023-07-28T13:42:51.497765Z","shell.execute_reply.started":"2023-07-28T13:42:51.488338Z","shell.execute_reply":"2023-07-28T13:42:51.495891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_terms = []\nfor aspect in ['BPO', 'CCO', 'MFO']:\n    selection = terms.loc[(terms.aspect == aspect)]\n    selection = selection.nlargest(V[aspect], columns='frequency', keep='first')\n    selected_terms += selection.term.to_list()\nselected_terms = set(selected_terms)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:42:51.499745Z","iopub.execute_input":"2023-07-28T13:42:51.500256Z","iopub.status.idle":"2023-07-28T13:42:51.548772Z","shell.execute_reply.started":"2023-07-28T13:42:51.500206Z","shell.execute_reply":"2023-07-28T13:42:51.547116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(selected_terms)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:42:51.551006Z","iopub.execute_input":"2023-07-28T13:42:51.55263Z","iopub.status.idle":"2023-07-28T13:42:51.563124Z","shell.execute_reply.started":"2023-07-28T13:42:51.552555Z","shell.execute_reply":"2023-07-28T13:42:51.561476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def assign_labels(annotations, selected_terms=selected_terms):\n    \n    intersection = selected_terms.intersection(annotations)\n    labels = np.isin(np.array(list(selected_terms)), np.array(list(intersection))) #+ random.rand(1)*0.000001 - random.rand(1)*0.000001 \n    \n    return list(labels.astype('int'))\n\nannotations = train_terms.groupby('EntryID')['term'].apply(set)\nlabels = annotations.progress_apply(assign_labels)\n\nlabels.head()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T13:42:51.565063Z","iopub.execute_input":"2023-07-28T13:42:51.566688Z","iopub.status.idle":"2023-07-28T13:46:56.977313Z","shell.execute_reply.started":"2023-07-28T13:42:51.566622Z","shell.execute_reply":"2023-07-28T13:46:56.976021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annotations","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:46:56.979414Z","iopub.execute_input":"2023-07-28T13:46:56.979813Z","iopub.status.idle":"2023-07-28T13:46:56.9953Z","shell.execute_reply.started":"2023-07-28T13:46:56.979775Z","shell.execute_reply":"2023-07-28T13:46:56.993794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:46:56.997474Z","iopub.execute_input":"2023-07-28T13:46:56.997914Z","iopub.status.idle":"2023-07-28T13:46:57.013809Z","shell.execute_reply.started":"2023-07-28T13:46:56.997871Z","shell.execute_reply":"2023-07-28T13:46:57.012027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading train embeddings","metadata":{}},{"cell_type":"markdown","source":"### K-mer Features\nThis is a popular algorithm used for converting protein sequences into numeric vectors. DNA sequence or protein sequence can be considered as a language. In K-mer algorithm, we split the long biological sentences into k-mer length overlapping words. For example, the sequence 'ATGCA' can be broken into 'ATG','TGC','GCA', if we fix the k-mer length equals 3. The kmer features can be then used to create sparse one-hot or tf-idf vectors, or dense vectors like embeddings. This is the same approach we call 'n-grams' or 'shingles' in natural language processing.","metadata":{}},{"cell_type":"code","source":"def kmer(seq,seq_length,kmer_length=3):\n    kmer_words = [seq[i:i+seq_length] for i in range(len(seq)-seq_length+1)]\n    print(kmer_words)\n    return ' '.join(kmer_words)\n\nkmer_length = 3\n#df['kmer_sequence'] = df['protein_sequence'].apply(lambda x: kmer(x,kmer_length))","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:46:57.01632Z","iopub.execute_input":"2023-07-28T13:46:57.016855Z","iopub.status.idle":"2023-07-28T13:46:57.025974Z","shell.execute_reply.started":"2023-07-28T13:46:57.016806Z","shell.execute_reply":"2023-07-28T13:46:57.024383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seq_test = 'ATGCA'\nkmer(seq_test, len(seq_test))","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:46:57.027809Z","iopub.execute_input":"2023-07-28T13:46:57.028648Z","iopub.status.idle":"2023-07-28T13:46:57.044768Z","shell.execute_reply.started":"2023-07-28T13:46:57.028593Z","shell.execute_reply":"2023-07-28T13:46:57.043469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:46:57.046167Z","iopub.execute_input":"2023-07-28T13:46:57.04674Z","iopub.status.idle":"2023-07-28T13:46:57.058522Z","shell.execute_reply.started":"2023-07-28T13:46:57.046697Z","shell.execute_reply":"2023-07-28T13:46:57.056688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/t5embeds/train_ids.npy\n# kaggle:  /kaggle/input/t5embeds/train_ids.npy\n\ntrain_ids = np.load('/kaggle/input/t5embeds/train_ids.npy') # TODO: see train_embeds\ny_train = np.array(labels[train_ids].to_list())","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T13:46:57.060894Z","iopub.execute_input":"2023-07-28T13:46:57.062667Z","iopub.status.idle":"2023-07-28T13:47:22.030168Z","shell.execute_reply.started":"2023-07-28T13:46:57.062595Z","shell.execute_reply":"2023-07-28T13:47:22.028448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:22.031836Z","iopub.execute_input":"2023-07-28T13:47:22.032368Z","iopub.status.idle":"2023-07-28T13:47:22.042767Z","shell.execute_reply.started":"2023-07-28T13:47:22.032318Z","shell.execute_reply":"2023-07-28T13:47:22.041101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#del labels\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:22.044512Z","iopub.execute_input":"2023-07-28T13:47:22.045238Z","iopub.status.idle":"2023-07-28T13:47:25.305937Z","shell.execute_reply.started":"2023-07-28T13:47:22.045189Z","shell.execute_reply":"2023-07-28T13:47:25.304712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/cafa-5-ems-2-embeddings-numpy/train_ids.npy\n# kaggle:  /kaggle/input/cafa-5-ems-2-embeddings-numpy/train_ids.npy \n\ntrain_ids2 = np.load('/kaggle/input/cafa-5-ems-2-embeddings-numpy/train_ids.npy')","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:25.307834Z","iopub.execute_input":"2023-07-28T13:47:25.308845Z","iopub.status.idle":"2023-07-28T13:47:25.393267Z","shell.execute_reply.started":"2023-07-28T13:47:25.308779Z","shell.execute_reply":"2023-07-28T13:47:25.392176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids.shape, train_ids2.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:25.395149Z","iopub.execute_input":"2023-07-28T13:47:25.39594Z","iopub.status.idle":"2023-07-28T13:47:25.403873Z","shell.execute_reply.started":"2023-07-28T13:47:25.395894Z","shell.execute_reply":"2023-07-28T13:47:25.402379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"A = {}\nShortList = []\nfor i in range(len(train_ids2)):\n    key = train_ids2[i]\n    A[key] = i\n        \nList = []\nfor i in range(len(train_ids)):\n    key = train_ids[i]\n    List.append(A[key])\n    if key in FullProt:\n        ShortList.append(i)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:25.405512Z","iopub.execute_input":"2023-07-28T13:47:25.405948Z","iopub.status.idle":"2023-07-28T13:47:25.915778Z","shell.execute_reply.started":"2023-07-28T13:47:25.405903Z","shell.execute_reply":"2023-07-28T13:47:25.913898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/protbert-embeddings-for-cafa5/train_embeddings.npy\n# kaggle:  ../input/protbert-embeddings-for-cafa5/train_embeddings.npy \n\nx_train3 = np.load('../input/protbert-embeddings-for-cafa5/train_embeddings.npy').astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:25.91746Z","iopub.execute_input":"2023-07-28T13:47:25.917897Z","iopub.status.idle":"2023-07-28T13:47:32.436634Z","shell.execute_reply.started":"2023-07-28T13:47:25.917859Z","shell.execute_reply":"2023-07-28T13:47:32.435155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train3 = x_train3[List,:]\nx_train3.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:32.445412Z","iopub.execute_input":"2023-07-28T13:47:32.44589Z","iopub.status.idle":"2023-07-28T13:47:32.715274Z","shell.execute_reply.started":"2023-07-28T13:47:32.445846Z","shell.execute_reply":"2023-07-28T13:47:32.713602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/cafa-5-ems-2-embeddings-numpy/train_embeddings.npy\n# kaggle:  ../input/cafa-5-ems-2-embeddings-numpy/train_embeddings.npy \n\nx_train2 = np.load('../input/cafa-5-ems-2-embeddings-numpy/train_embeddings.npy').astype(np.float32)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T13:47:32.716726Z","iopub.execute_input":"2023-07-28T13:47:32.717163Z","iopub.status.idle":"2023-07-28T13:47:40.107848Z","shell.execute_reply.started":"2023-07-28T13:47:32.717106Z","shell.execute_reply":"2023-07-28T13:47:40.106083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train2 = x_train2[List,:]\nx_train2.shape","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T13:47:40.109845Z","iopub.execute_input":"2023-07-28T13:47:40.110912Z","iopub.status.idle":"2023-07-28T13:47:40.424611Z","shell.execute_reply.started":"2023-07-28T13:47:40.110862Z","shell.execute_reply":"2023-07-28T13:47:40.423268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/t5embeds/train_embeds.npy\n# kaggle:  /kaggle/input/t5embeds/train_embeds.npy  \n\nx_train = np.load('/kaggle/input/t5embeds/train_embeds.npy').astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:40.426447Z","iopub.execute_input":"2023-07-28T13:47:40.427333Z","iopub.status.idle":"2023-07-28T13:47:52.780313Z","shell.execute_reply.started":"2023-07-28T13:47:40.427275Z","shell.execute_reply":"2023-07-28T13:47:52.778692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/cafa5-amino-counts/train_amino_counts.csv\n# kaggle:  /kaggle/input/cafa5-amino-counts/train_amino_counts.csv\n\nx_train_amino_counts4 = pd.read_csv('/kaggle/input/cafa5-amino-counts/train_amino_counts.csv' ).to_numpy().astype(np.float32)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T13:47:52.782006Z","iopub.execute_input":"2023-07-28T13:47:52.782566Z","iopub.status.idle":"2023-07-28T13:47:53.353204Z","shell.execute_reply.started":"2023-07-28T13:47:52.782507Z","shell.execute_reply":"2023-07-28T13:47:53.351233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train,x_train2,x_train3,x_train_amino_counts4","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:53.355263Z","iopub.execute_input":"2023-07-28T13:47:53.355723Z","iopub.status.idle":"2023-07-28T13:47:53.36969Z","shell.execute_reply.started":"2023-07-28T13:47:53.355678Z","shell.execute_reply":"2023-07-28T13:47:53.368023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x_train = np.concatenate([x_train,x_train2,x_train3,x_train_amino_counts4], axis =1)\n# remove prot_bert x3 - score 0.47263\n# remove ems2 x2 - score \n# remove ems2 x2 - score \n# remove t5 x_train - score\nx_train = np.concatenate([x_train,x_train2,x_train3], axis =1)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:53.371798Z","iopub.execute_input":"2023-07-28T13:47:53.37238Z","iopub.status.idle":"2023-07-28T13:47:54.318688Z","shell.execute_reply.started":"2023-07-28T13:47:53.372319Z","shell.execute_reply":"2023-07-28T13:47:54.316919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = x_train[ShortList,:]","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:54.320209Z","iopub.execute_input":"2023-07-28T13:47:54.320622Z","iopub.status.idle":"2023-07-28T13:47:54.625047Z","shell.execute_reply.started":"2023-07-28T13:47:54.320584Z","shell.execute_reply":"2023-07-28T13:47:54.623208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = y_train[ShortList,:]","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:54.626691Z","iopub.execute_input":"2023-07-28T13:47:54.62714Z","iopub.status.idle":"2023-07-28T13:47:54.893356Z","shell.execute_reply.started":"2023-07-28T13:47:54.627078Z","shell.execute_reply":"2023-07-28T13:47:54.891367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del x_train3\ndel x_train2\ndel annotations\ndel train_terms\ndel terms\ndel train_ids\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:54.895615Z","iopub.execute_input":"2023-07-28T13:47:54.89732Z","iopub.status.idle":"2023-07-28T13:47:58.461684Z","shell.execute_reply.started":"2023-07-28T13:47:54.897252Z","shell.execute_reply":"2023-07-28T13:47:58.45982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape, y_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:47:58.463868Z","iopub.execute_input":"2023-07-28T13:47:58.464472Z","iopub.status.idle":"2023-07-28T13:47:58.47607Z","shell.execute_reply.started":"2023-07-28T13:47:58.464404Z","shell.execute_reply":"2023-07-28T13:47:58.474131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Ridge","metadata":{}},{"cell_type":"code","source":"model = Ridge(alpha = 10, random_state=42).fit(x_train, y_train)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T13:47:58.478954Z","iopub.execute_input":"2023-07-28T13:47:58.479903Z","iopub.status.idle":"2023-07-28T13:48:12.772393Z","shell.execute_reply.started":"2023-07-28T13:47:58.479829Z","shell.execute_reply":"2023-07-28T13:48:12.769811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#del x_train, y_train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:48:12.7773Z","iopub.execute_input":"2023-07-28T13:48:12.781392Z","iopub.status.idle":"2023-07-28T13:48:16.719046Z","shell.execute_reply.started":"2023-07-28T13:48:12.781165Z","shell.execute_reply":"2023-07-28T13:48:16.71686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training CatBoost","metadata":{}},{"cell_type":"code","source":"#x_train, x_valid, y_train, y_valid = train_test_split(x_train, y_train, shuffle=True, random_state=42)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T13:48:16.721876Z","iopub.execute_input":"2023-07-28T13:48:16.722449Z","iopub.status.idle":"2023-07-28T13:48:16.738128Z","shell.execute_reply.started":"2023-07-28T13:48:16.722399Z","shell.execute_reply":"2023-07-28T13:48:16.736158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model_CatB = CatBoostRegressor( loss_function= 'MultiRMSE', eval_metric='MultiRMSE', task_type= 'CPU' )\n#model_CatB.fit(X = x_train , y= y_train , eval_set=(x_valid , y_valid ), verbose=False)  ","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T13:48:16.739931Z","iopub.execute_input":"2023-07-28T13:48:16.740409Z","iopub.status.idle":"2023-07-28T13:48:16.753856Z","shell.execute_reply.started":"2023-07-28T13:48:16.740364Z","shell.execute_reply":"2023-07-28T13:48:16.752064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#y_train","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:48:16.756313Z","iopub.execute_input":"2023-07-28T13:48:16.756959Z","iopub.status.idle":"2023-07-28T13:48:16.774983Z","shell.execute_reply.started":"2023-07-28T13:48:16.756894Z","shell.execute_reply":"2023-07-28T13:48:16.773206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#labels[train_ids].to_list()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T13:48:16.777442Z","iopub.execute_input":"2023-07-28T13:48:16.778039Z","iopub.status.idle":"2023-07-28T13:48:16.790848Z","shell.execute_reply.started":"2023-07-28T13:48:16.777972Z","shell.execute_reply":"2023-07-28T13:48:16.789452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training mlp  from notebook https://www.kaggle.com/code/simonveitner/simple-mlp","metadata":{}},{"cell_type":"code","source":"x_train, x_valid, y_train, y_valid = train_test_split(x_train, y_train, shuffle=True, random_state=42)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T13:48:16.793918Z","iopub.execute_input":"2023-07-28T13:48:16.794586Z","iopub.status.idle":"2023-07-28T13:48:17.418268Z","shell.execute_reply.started":"2023-07-28T13:48:16.794522Z","shell.execute_reply":"2023-07-28T13:48:17.416919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build a simple MLP model in Keras with ReLU activation and nothing else\nnfeats = x_train.shape[1]\nnlabels = y_train.shape[1]\nearly_stopping = callbacks.EarlyStopping(\n    min_delta=0.001, # minimium amount of change to count as an improvement\n    patience=5, # how many epochs to wait before stopping\n    restore_best_weights=True,\n)\nmodel_mlp = keras.Sequential([\n    layers.BatchNormalization(input_shape=[nfeats]),\n    layers.Dense(2048, activation=\"swish\"),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(2048, activation=\"swish\"),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(2048, activation=\"swish\"),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(2048, activation=\"swish\"),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(2048, activation=\"swish\"),\n    layers.BatchNormalization(),\n    layers.Dropout(0.3),\n    layers.Dense(nlabels, activation='sigmoid'),\n])\n#model_mlp = Sequential()\n#model_mlp.add(Dense(3100, activation='relu', input_dim=nfeats))\n#model_mlp.add(Dropout(0.7))\n#model_mlp.add(Dense(2500 , activation='relu'))\n#model_mlp.add(Dense(2150, activation='relu'))\n#model_mlp.add(Dense(128, activation='relu'))\n#model_mlp.add(Dense(nlabels, activation='sigmoid'))\nmodel_mlp.compile(loss='binary_crossentropy',\n                optimizer='adam',\n                metrics=[AUC()])\nmodel_mlp.summary()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T13:48:17.420134Z","iopub.execute_input":"2023-07-28T13:48:17.421737Z","iopub.status.idle":"2023-07-28T13:48:18.034304Z","shell.execute_reply.started":"2023-07-28T13:48:17.421669Z","shell.execute_reply":"2023-07-28T13:48:18.031956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:48:18.036761Z","iopub.execute_input":"2023-07-28T13:48:18.037261Z","iopub.status.idle":"2023-07-28T13:48:21.491501Z","shell.execute_reply.started":"2023-07-28T13:48:18.037214Z","shell.execute_reply":"2023-07-28T13:48:21.489499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#   11   0.9442  2700  1850\n#model_mlp.fit(x_train, y_train, epochs=30, batch_size=128)#, validation_data=(x_valid, y_valid))\nhistory = model_mlp.fit(\n    x_train, y_train,\n    validation_data=(x_valid, y_valid),\n    batch_size=512,\n    epochs=50,\n    callbacks=[early_stopping], # put your callbacks in a list\n    verbose=0,  # turn off training log\n)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T13:48:21.493763Z","iopub.execute_input":"2023-07-28T13:48:21.494345Z","iopub.status.idle":"2023-07-28T14:07:25.079374Z","shell.execute_reply.started":"2023-07-28T13:48:21.494296Z","shell.execute_reply":"2023-07-28T14:07:25.076192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df = pd.DataFrame(history.history)\nhistory_df.loc[:, ['loss', 'val_loss']].plot();\nprint(\"Minimum validation loss: {}\".format(history_df['val_loss'].min()))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_df.loc[:,['auc', 'val_auc']].plot()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/t5embeds/test_ids.npy\n#          /home/datalab/nfs/competitionCAFA5/kaggle/t5embeds/test_embeds.npy\n# kaggle:  /kaggle/input/t5embeds/test_ids.npy\n#          /kaggle/input/t5embeds/test_embeds.npy   \n\ntest_ids = np.load('/kaggle/input/t5embeds/test_ids.npy')\nx_test = np.load('/kaggle/input/t5embeds/test_embeds.npy').astype(np.float32)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:47:39.754934Z","iopub.execute_input":"2023-07-28T10:47:39.756336Z","iopub.status.idle":"2023-07-28T10:47:51.761771Z","shell.execute_reply.started":"2023-07-28T10:47:39.756279Z","shell.execute_reply":"2023-07-28T10:47:51.759804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/cafa-5-ems-2-embeddings-numpy/test_ids.npy\n# kaggle:  /kaggle/input/cafa-5-ems-2-embeddings-numpy/test_ids.npy \n\ntest_ids2 = np.load('/kaggle/input/cafa-5-ems-2-embeddings-numpy/test_ids.npy')","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:47:59.882178Z","iopub.execute_input":"2023-07-28T10:47:59.883058Z","iopub.status.idle":"2023-07-28T10:47:59.974812Z","shell.execute_reply.started":"2023-07-28T10:47:59.883008Z","shell.execute_reply":"2023-07-28T10:47:59.973104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/protbert-embeddings-for-cafa5/test_ids.npy\n# kaggle:  /kaggle/input/protbert-embeddings-for-cafa5/test_ids.npy\n\n#order as test_ids\ntest_ids3 = np.load('/kaggle/input/protbert-embeddings-for-cafa5/test_ids.npy')","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:48:02.383631Z","iopub.execute_input":"2023-07-28T10:48:02.384568Z","iopub.status.idle":"2023-07-28T10:48:02.453496Z","shell.execute_reply.started":"2023-07-28T10:48:02.384496Z","shell.execute_reply":"2023-07-28T10:48:02.451961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/cafa5-amino-counts/test_amino_counts.csv\n# kaggle:  /kaggle/input/cafa5-amino-counts/test_amino_counts.csv\n# No id. keep as test_ids\n#4","metadata":{"execution":{"iopub.status.busy":"2023-07-09T18:11:30.759368Z","iopub.execute_input":"2023-07-09T18:11:30.759801Z","iopub.status.idle":"2023-07-09T18:11:30.76617Z","shell.execute_reply.started":"2023-07-09T18:11:30.759758Z","shell.execute_reply":"2023-07-09T18:11:30.764499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:48:06.767281Z","iopub.execute_input":"2023-07-28T10:48:06.768489Z","iopub.status.idle":"2023-07-28T10:48:06.778943Z","shell.execute_reply.started":"2023-07-28T10:48:06.768426Z","shell.execute_reply":"2023-07-28T10:48:06.777447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids2 ","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:48:09.926529Z","iopub.execute_input":"2023-07-28T10:48:09.926964Z","iopub.status.idle":"2023-07-28T10:48:09.935801Z","shell.execute_reply.started":"2023-07-28T10:48:09.926926Z","shell.execute_reply":"2023-07-28T10:48:09.934084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids3","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:48:13.189477Z","iopub.execute_input":"2023-07-28T10:48:13.18991Z","iopub.status.idle":"2023-07-28T10:48:13.199603Z","shell.execute_reply.started":"2023-07-28T10:48:13.189871Z","shell.execute_reply":"2023-07-28T10:48:13.198202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids.shape, test_ids2.shape, test_ids3.shape","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:48:16.761517Z","iopub.execute_input":"2023-07-28T10:48:16.761958Z","iopub.status.idle":"2023-07-28T10:48:16.773254Z","shell.execute_reply.started":"2023-07-28T10:48:16.761918Z","shell.execute_reply":"2023-07-28T10:48:16.771735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"A = {}\nfor i in range(len(test_ids2)):\n    key = test_ids2[i]\n    A[key] = i\nList = []\nfor i in range(len(test_ids)):\n    key = test_ids[i]\n    List.append(A[key])\nList[0]","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:48:19.390898Z","iopub.execute_input":"2023-07-28T10:48:19.391356Z","iopub.status.idle":"2023-07-28T10:48:19.913493Z","shell.execute_reply.started":"2023-07-28T10:48:19.391318Z","shell.execute_reply":"2023-07-28T10:48:19.911856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids[7], List[7], test_ids2[List[7]] , test_ids3[7]#, test_amino_counts4[List[7]]","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:48:24.52337Z","iopub.execute_input":"2023-07-28T10:48:24.523823Z","iopub.status.idle":"2023-07-28T10:48:24.53416Z","shell.execute_reply.started":"2023-07-28T10:48:24.523784Z","shell.execute_reply":"2023-07-28T10:48:24.532458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/cafa-5-ems-2-embeddings-numpy/test_embeddings.npy \n# kaggle:  /kaggle/input/cafa-5-ems-2-embeddings-numpy/test_embeddings.npy\n\nx_test2 = np.load('/kaggle/input/cafa-5-ems-2-embeddings-numpy/test_embeddings.npy').astype(np.float32)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:48:29.08644Z","iopub.execute_input":"2023-07-28T10:48:29.086892Z","iopub.status.idle":"2023-07-28T10:48:39.638469Z","shell.execute_reply.started":"2023-07-28T10:48:29.086851Z","shell.execute_reply":"2023-07-28T10:48:39.63703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test2 = x_test2[List,:]\nx_test2.shape","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:48:54.704474Z","iopub.execute_input":"2023-07-28T10:48:54.70492Z","iopub.status.idle":"2023-07-28T10:48:55.012746Z","shell.execute_reply.started":"2023-07-28T10:48:54.704881Z","shell.execute_reply":"2023-07-28T10:48:55.011404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/protbert-embeddings-for-cafa5/test_embeddings.npy \n# kaggle:  /kaggle/input/protbert-embeddings-for-cafa5/test_embeddings.npy\n\nx_test3 = np.load('/kaggle/input/protbert-embeddings-for-cafa5/test_embeddings.npy').astype(np.float32)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:48:57.627315Z","iopub.execute_input":"2023-07-28T10:48:57.627778Z","iopub.status.idle":"2023-07-28T10:49:03.668711Z","shell.execute_reply.started":"2023-07-28T10:48:57.627737Z","shell.execute_reply":"2023-07-28T10:49:03.667429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test3.shape","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:49:06.378829Z","iopub.execute_input":"2023-07-28T10:49:06.379363Z","iopub.status.idle":"2023-07-28T10:49:06.389268Z","shell.execute_reply.started":"2023-07-28T10:49:06.379313Z","shell.execute_reply":"2023-07-28T10:49:06.387799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# local:   /home/datalab/nfs/competitionCAFA5/kaggle/cafa5-amino-counts/test_amino_counts.csv\n# kaggle:  /kaggle/input/cafa5-amino-counts/test_amino_counts.csv\n\nx_test_amino_counts4 = pd.read_csv('/kaggle/input/cafa5-amino-counts/test_amino_counts.csv').to_numpy().astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:49:12.239454Z","iopub.execute_input":"2023-07-28T10:49:12.239959Z","iopub.status.idle":"2023-07-28T10:49:12.767416Z","shell.execute_reply.started":"2023-07-28T10:49:12.239909Z","shell.execute_reply":"2023-07-28T10:49:12.765972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test_amino_counts4.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-28T10:49:19.727734Z","iopub.execute_input":"2023-07-28T10:49:19.728405Z","iopub.status.idle":"2023-07-28T10:49:19.737469Z","shell.execute_reply.started":"2023-07-28T10:49:19.728343Z","shell.execute_reply":"2023-07-28T10:49:19.735896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x_test = np.concatenate([x_test,x_test2,x_test3,x_test_amino_counts4], axis =1)\nx_test = np.concatenate([x_test,x_test2,x_test3], axis =1)\nx_test.shape","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:49:22.365633Z","iopub.execute_input":"2023-07-28T10:49:22.367308Z","iopub.status.idle":"2023-07-28T10:49:23.378228Z","shell.execute_reply.started":"2023-07-28T10:49:22.367251Z","shell.execute_reply":"2023-07-28T10:49:23.37646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del x_test2\ndel x_test3\ndel x_test_amino_counts4\n\ngc.collect()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:49:26.654904Z","iopub.execute_input":"2023-07-28T10:49:26.656056Z","iopub.status.idle":"2023-07-28T10:49:52.514484Z","shell.execute_reply.started":"2023-07-28T10:49:26.655971Z","shell.execute_reply":"2023-07-28T10:49:52.512499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# predictions Ridge ","metadata":{"execution":{"iopub.execute_input":"2023-07-08T14:56:20.788398Z","iopub.status.busy":"2023-07-08T14:56:20.787929Z","iopub.status.idle":"2023-07-08T14:56:20.791401Z","shell.execute_reply":"2023-07-08T14:56:20.790846Z","shell.execute_reply.started":"2023-07-08T14:56:20.788378Z"}}},{"cell_type":"code","source":"del model","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\npredictions = model.predict(x_test)\n#del x_test\ndel model\ngc.collect()\nprint(type(predictions[0,0]))\npredictions = predictions.astype(np.float16)\nsub = pd.DataFrame(data=predictions, columns=selected_terms, index=test_ids)\n\nprint(predictions.shape)\n\ndel predictions\ngc.collect()\n\nsub = sub.T.unstack().reset_index(name='prediction')\nsub = sub.loc[sub['prediction'] > 0.1]\nsub.head()\n\n##### local:   /home/datalab/nfs/competitionCAFA5/output/combineembeddings/submission.tsv \n##### kaggle:  submission.tsv\n\nsub.to_csv('submission.tsv', sep='\\t', index=False, header=False)\n\ndel sub\ngc.collect()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-09T18:11:31.340709Z","iopub.status.idle":"2023-07-09T18:11:31.341375Z","shell.execute_reply.started":"2023-07-09T18:11:31.341023Z","shell.execute_reply":"2023-07-09T18:11:31.341058Z"}}},{"cell_type":"markdown","source":"# predictions_MLP (== 15 batch)","metadata":{}},{"cell_type":"code","source":"batch_size=10000 # total: 141865\n\nfor index in range(0,x_test.shape[0],batch_size):\n    batch=x_test[index:min(index+batch_size,x_test.shape[0]),:]\n    batch_ids=test_ids[index:min(index+batch_size,test_ids.shape[0])]\n    print(f\"index={index}, {batch.shape}  {batch_ids.shape}:\")\n    \n    predictions_MLP = model_mlp.predict(batch)\n\n    del batch\n    gc.collect()\n\n    print(f\"   {type(predictions_MLP[0,0])}\")\n\n    predictions_MLP = predictions_MLP.astype(np.float16)\n\n    sub = pd.DataFrame(data=predictions_MLP, columns=selected_terms, index=batch_ids)\n    print(f\"   {predictions_MLP.shape}\")\n\n    del predictions_MLP\n    gc.collect()\n\n    sub = sub.T.unstack().reset_index(name='prediction')\n    sub = sub.loc[sub['prediction'] > 0]\n    sub.head()\n\n    # local:   /home/datalab/nfs/competitionCAFA5/output/combineembeddings/submission.tsv \n    # kaggle:  submission.tsv\n\n    if index==0:\n        mode_value ='w'\n    else: \n        mode_value ='a'        \n    sub.to_csv('submission.tsv', sep='\\t', index=False, header=False, mode=mode_value)\ndel model_mlp   ","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-07-28T10:49:56.327933Z","iopub.execute_input":"2023-07-28T10:49:56.328415Z","iopub.status.idle":"2023-07-28T11:10:30.22412Z","shell.execute_reply.started":"2023-07-28T10:49:56.32837Z","shell.execute_reply":"2023-07-28T11:10:30.222544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-28T14:49:39.782218Z","iopub.execute_input":"2023-07-28T14:49:39.782951Z","iopub.status.idle":"2023-07-28T14:49:40.315977Z","shell.execute_reply.started":"2023-07-28T14:49:39.782862Z","shell.execute_reply":"2023-07-28T14:49:40.313225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ENSEMBLING WITH TOP PUBLIC SUBMISSION","metadata":{}},{"cell_type":"code","source":"# local:    /home/datalab/nfs/competitionCAFA5/kaggle/cafa-5-053818-pred/submission (3).tsv\n#           /home/datalab/nfs/competitionCAFA5/kaggle/foldseek-cafa/foldseek_submission.tsv\n#           /home/datalab/nfs/competitionCAFA5/kaggle/cafa5-tuning-merge-datasets/submission.tsv\n#           /home/datalab/nfs/competitionCAFA5/output/submission.tsv  \n# kaggle:   /kaggle/input/cafa-5-053818-pred/submission (3).tsv\n#           /kaggle/input/foldseek-cafa/foldseek_submission.tsv\n#           /kaggle/input/cafa5-tuning-merge-datasets/submission.tsv\n#           submission.tsv\n\n#submission_best_public2 = pd.read_csv('/kaggle/input/cafa-5-053818-pred/submission (3).tsv', sep='\\t', header=None, names=['Id2', 'GO term2', 'Confidence2'])\n\n#test_pred_df_foldseek = pd.read_csv('/kaggle/input/foldseek-cafa/foldseek_submission.tsv', sep='\\t', header=None, names=[1, 2, 3])\n#test_pred_df_foldseek = test_pred_df_foldseek[test_pred_df_foldseek[3] > 0.6]\n\n#submission_best_public = pd.read_csv('/kaggle/input/cafa5-tuning-merge-datasets/submission.tsv', sep='\\t', header=None, names=['Id', 'GO term', 'Confidence'])\n\n#submissions_merged = submission_best_public.merge(test_pred_df_foldseek, left_on=['Id', 'GO term'], \n#                                                  right_on=[1, 2], how='outer')\n#submissions_merged.drop([1, 2], axis=1, inplace=True)\n#submissions_merged['confidence_combined'] = submissions_merged.apply(lambda row: row['Confidence'] if not np.isnan(row['Confidence']) else row[3], axis=1)\n\n#submissions_merged[['Id', 'GO term', 'confidence_combined']].to_csv('submission.tsv', sep='\\t', header=False, index=False)","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]}]}