{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ? \n\nSimple Baseline to start with : Covert MultiLabel to MultiTarge  + Embeddings + Ridge \n\n    Features - precalculated embeddings for protein sequences. Thanks to Grandmaster Sergei Fironov for sharing protein emebedding calculated by T5 protein language model from the Rost Lab. \n    \n    Targets - multi-label is converted to mult-target (binary classification) task - i.e. for each sample we are preciting the probability that this label is assigned to that sample. In total there can be 40 000 labels - that is too much, so we choose only N the most frequent ones. \n    \n    After that - use any ML-model you like to make predictions. Start with Ridge as the he most simple and fast one. \n    ","metadata":{}},{"cell_type":"markdown","source":"Thanks to all  authors of the public notebooks and datasets which are quite helpful (please upvote them) and especially those ones:\n\nLEONID KULYK: https://www.kaggle.com/code/leonidkulyk/eda-cafa5-pfp-interactive-dags-plotly\n\nMARÍLIA PRATA: https://www.kaggle.com/code/mpwolke/cafa-5-protein-prediction\n\nDAREK KŁECZEK:  https://www.kaggle.com/code/thedrcat/cafa-eda\n\nD_KHATRI:  https://www.kaggle.com/code/dhruvkhatri/naive-submission-afa\n\n* Pretrained T5 protein embeddings: \n    * https://www.kaggle.com/datasets/danofer/uniprotkbswiss-prot-protein-embeddings\n\nGrandmaster Sergei Fironov shared protein emebedding calculated by T5 protein language model from the Rost Lab:  https://www.kaggle.com/datasets/sergeifironov/t5embeds\n\n","metadata":{}},{"cell_type":"markdown","source":"# Key param(s)\n\n","metadata":{}},{"cell_type":"code","source":"n_labels_to_consider = 1499 # We will choose only top frequent labels (in train) and predict only them. \nn_max_preds = 1499","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:28:58.194157Z","iopub.execute_input":"2023-04-25T11:28:58.194669Z","iopub.status.idle":"2023-04-25T11:28:58.200751Z","shell.execute_reply.started":"2023-04-25T11:28:58.194625Z","shell.execute_reply":"2023-04-25T11:28:58.198881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nt0start = time.time() \n\nimport numpy as np\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import Ridge,RidgeCV\nfrom sklearn.neural_network import MLPClassifier\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-25T10:19:50.183647Z","iopub.execute_input":"2023-04-25T10:19:50.184244Z","iopub.status.idle":"2023-04-25T10:19:51.561004Z","shell.execute_reply.started":"2023-04-25T10:19:50.184204Z","shell.execute_reply":"2023-04-25T10:19:51.559623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare multi-target Y  ( transition from multi-label task to multi target task - binary classifiction ). ","metadata":{}},{"cell_type":"markdown","source":"## Load train labels and select the most frequent ones","metadata":{}},{"cell_type":"code","source":"%%time\ntrainTerms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\",sep=\"\\t\")\nprint(trainTerms.shape)\ndisplay(trainTerms.head(2))\nvec_freqCount = (trainTerms['term'].value_counts())\nprint(vec_freqCount )","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:19:51.562635Z","iopub.execute_input":"2023-04-25T10:19:51.56492Z","iopub.status.idle":"2023-04-25T10:19:56.264881Z","shell.execute_reply.started":"2023-04-25T10:19:51.564859Z","shell.execute_reply":"2023-04-25T10:19:56.263371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## drop very rares\nvec_freqCount = vec_freqCount[vec_freqCount>=30]\nprint(vec_freqCount.shape[0])\nvec_freqCount.describe().round()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:19:56.269384Z","iopub.execute_input":"2023-04-25T10:19:56.26983Z","iopub.status.idle":"2023-04-25T10:19:56.287304Z","shell.execute_reply.started":"2023-04-25T10:19:56.269788Z","shell.execute_reply":"2023-04-25T10:19:56.285743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vec_freqCount[vec_freqCount>200].shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:19:56.288978Z","iopub.execute_input":"2023-04-25T10:19:56.289587Z","iopub.status.idle":"2023-04-25T10:19:56.298792Z","shell.execute_reply.started":"2023-04-25T10:19:56.289545Z","shell.execute_reply":"2023-04-25T10:19:56.297477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print()\nlabels_to_consider = list(vec_freqCount.index[:n_labels_to_consider] )\nprint('n_labels_to_consider:', len(labels_to_consider), 'First 10:', labels_to_consider[:10] ) ","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:19:56.300977Z","iopub.execute_input":"2023-04-25T10:19:56.301467Z","iopub.status.idle":"2023-04-25T10:19:56.31122Z","shell.execute_reply.started":"2023-04-25T10:19:56.301415Z","shell.execute_reply":"2023-04-25T10:19:56.309739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load protein Ids in train","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/train_ids.npy'\nvec_train_protein_ids = np.load(fn)\nprint(vec_train_protein_ids.shape)\nvec_train_protein_ids","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:19:56.312948Z","iopub.execute_input":"2023-04-25T10:19:56.313484Z","iopub.status.idle":"2023-04-25T10:19:56.388624Z","shell.execute_reply.started":"2023-04-25T10:19:56.313439Z","shell.execute_reply":"2023-04-25T10:19:56.387588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Y ","metadata":{}},{"cell_type":"code","source":"%%time \ntrain_size = 142246 # len(X)\nY = np.zeros( (train_size ,n_labels_to_consider) )\nprint(Y.shape)\n\nseries_train_protein_ids = pd.Series(vec_train_protein_ids ) # \n\ntrainTerms_smaller = trainTerms[ trainTerms['term'].isin( labels_to_consider ) ] # to speed-up the next step \nprint( trainTerms_smaller.shape)\n\nfor i in range(Y.shape[1]):\n    m = trainTerms_smaller['term'] ==  labels_to_consider[i]\n#     m.sum()\n    Y[:,i] =  series_train_protein_ids.isin(  set(trainTerms_smaller[m]['EntryID'] ) ).astype(float )\n    if (i % 10) == 0: \n        print(i, m.sum())\nY ","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:19:56.389686Z","iopub.execute_input":"2023-04-25T10:19:56.390948Z","iopub.status.idle":"2023-04-25T10:27:55.151791Z","shell.execute_reply.started":"2023-04-25T10:19:56.390901Z","shell.execute_reply":"2023-04-25T10:27:55.150345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n# save for possible future reuse \nfn4saveY = 'Y_'+str(Y.shape[1])\nprint(fn4saveY)\nnp.save( fn4saveY , Y) ","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:27:55.153782Z","iopub.execute_input":"2023-04-25T10:27:55.154803Z","iopub.status.idle":"2023-04-25T10:27:56.688102Z","shell.execute_reply.started":"2023-04-25T10:27:55.154745Z","shell.execute_reply":"2023-04-25T10:27:56.686599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfn4save_labels = 'Y_'+str(Y.shape[1]) + '_labels'\nnp.save(fn4save_labels, labels_to_consider )","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:27:56.693739Z","iopub.execute_input":"2023-04-25T10:27:56.694134Z","iopub.status.idle":"2023-04-25T10:27:56.703395Z","shell.execute_reply.started":"2023-04-25T10:27:56.694097Z","shell.execute_reply":"2023-04-25T10:27:56.701808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print( list(np.load(fn4save_labels +'.npy' ))[:10] )","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:27:56.704881Z","iopub.execute_input":"2023-04-25T10:27:56.70526Z","iopub.status.idle":"2023-04-25T10:27:56.711859Z","shell.execute_reply.started":"2023-04-25T10:27:56.705223Z","shell.execute_reply":"2023-04-25T10:27:56.710418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n# Someone may prefer  Y as dataframe \nif 1:\n    df_Y = pd.DataFrame(data = Y, columns = labels_to_consider)\n    display(df_Y.head(2))\n#     print( df.info().sum() )\n    print('memory_usage:', df_Y.memory_usage(index=True).sum() )\n    display(df_Y.describe() )    \n    fn4save =  'df_Y_'+str(Y.shape[1]) + '.csv'\n    df_Y.to_csv(fn4save)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:27:56.713515Z","iopub.execute_input":"2023-04-25T10:27:56.714369Z","iopub.status.idle":"2023-04-25T10:29:51.963323Z","shell.execute_reply.started":"2023-04-25T10:27:56.714269Z","shell.execute_reply":"2023-04-25T10:29:51.961941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load train features - precalculated embeddings for the proteins","metadata":{}},{"cell_type":"code","source":"%%time\n\n# fn = '/kaggle/input/protein-embeddings-1/reduced_embeddings_file.npy'\n# fn = '/kaggle/input/protein-embeddings-1/embed_protbert_train_clip_1200_first_70000_prot.csv'\nfn = '/kaggle/input/t5embeds/train_embeds.npy'\n# fn = '/kaggle/input/t5embeds/test_embeds.npy'\n\nprint(fn)\nif '.csv' in fn:\n    df = pd.read_csv(fn, index_col = 0)\n    X = df.values\nelif '.npy' in fn:\n    X = np.load(fn)\nprint(X.shape)\nX","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:29:51.965058Z","iopub.execute_input":"2023-04-25T10:29:51.965447Z","iopub.status.idle":"2023-04-25T10:30:03.369128Z","shell.execute_reply.started":"2023-04-25T10:29:51.965402Z","shell.execute_reply":"2023-04-25T10:30:03.367853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load protein Ids ","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/train_ids.npy'\nvec_train_protein_ids = np.load(fn)\nprint(vec_train_protein_ids.shape)\nvec_train_protein_ids","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:30:03.370537Z","iopub.execute_input":"2023-04-25T10:30:03.371114Z","iopub.status.idle":"2023-04-25T10:30:03.385077Z","shell.execute_reply.started":"2023-04-25T10:30:03.371066Z","shell.execute_reply":"2023-04-25T10:30:03.383767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sanity check \n\nIds from the train data are the same as from the train labels data","metadata":{}},{"cell_type":"code","source":"s = set(vec_train_protein_ids) &set (trainTerms['EntryID'] )\nprint( len(s), len( X ) )  # get same numbers ","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:30:03.387275Z","iopub.execute_input":"2023-04-25T10:30:03.387779Z","iopub.status.idle":"2023-04-25T10:30:04.046983Z","shell.execute_reply.started":"2023-04-25T10:30:03.387725Z","shell.execute_reply":"2023-04-25T10:30:04.045614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare Train-Test split ","metadata":{"execution":{"iopub.status.busy":"2023-04-24T10:45:13.246257Z","iopub.execute_input":"2023-04-24T10:45:13.247139Z","iopub.status.idle":"2023-04-24T10:45:23.612468Z","shell.execute_reply.started":"2023-04-24T10:45:13.247095Z","shell.execute_reply":"2023-04-24T10:45:23.611491Z"}}},{"cell_type":"code","source":"IX = np.arange(len(X))\nIX_train, IX_test, _,_ = train_test_split( IX, IX, train_size=0.1, random_state=42)\nprint(len(IX_train), len(IX_test),  IX_train[:10], IX_test[:10] )","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:30:04.048901Z","iopub.execute_input":"2023-04-25T10:30:04.049383Z","iopub.status.idle":"2023-04-25T10:30:04.067202Z","shell.execute_reply.started":"2023-04-25T10:30:04.049328Z","shell.execute_reply":"2023-04-25T10:30:04.066081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier\n\n# model = Ridge(alpha=1.0) # 0.805\nmodel = RidgeCV() # 0.8127 auc, with 15% train\n# model = RandomForestClassifier(n_estimators=200,  max_depth=14, min_samples_split=3, min_samples_leaf=1,n_jobs=-1) ## much slower... \n# model =MLPClassifier(hidden_layer_sizes=(512,256), early_stopping=True,\n#                      validation_fraction=0.05,learning_rate=\"adaptive\",learning_rate_init=0.005) # 0.59 rocauc , and slower\nstr_model_id = 'Ridge1'\n\ndf_models_stat = pd.DataFrame()\nmodel","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:30:04.069016Z","iopub.execute_input":"2023-04-25T10:30:04.069514Z","iopub.status.idle":"2023-04-25T10:30:04.382937Z","shell.execute_reply.started":"2023-04-25T10:30:04.069464Z","shell.execute_reply":"2023-04-25T10:30:04.381544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nimport time\nfrom sklearn.metrics import roc_auc_score\n\nt0 = time.time()\nmodel.fit(X[IX_train,:],Y[IX_train,:])\nY_pred_test = model.predict(X[IX_test,:])\ntt = time.time() - t0\nprint(str_model_id, tt)\nl = []\nfor i in range(Y.shape[1]):\n    if len(np.unique(Y[IX_test,i]) ) > 1:\n        s = roc_auc_score(Y[IX_test,i], Y_pred_test[:,i]);\n    else:\n        s = 0.5\n    l.append(s)        \n    if i %10 == 0:\n        print(i, s)\ndf_models_stat.loc[str_model_id,'RocAuc Mean Test'] = np.mean(l)\ndf_models_stat.loc[str_model_id,'Time'] = np.round(tt,1)\ndf_models_stat.loc[str_model_id,'Test Size'] = len(IX_test)\ndf_models_stat","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:30:04.384596Z","iopub.execute_input":"2023-04-25T10:30:04.384986Z","iopub.status.idle":"2023-04-25T10:31:54.323276Z","shell.execute_reply.started":"2023-04-25T10:30:04.384945Z","shell.execute_reply":"2023-04-25T10:31:54.321486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.get_params()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:31:54.325721Z","iopub.execute_input":"2023-04-25T10:31:54.326117Z","iopub.status.idle":"2023-04-25T10:31:54.336364Z","shell.execute_reply.started":"2023-04-25T10:31:54.326078Z","shell.execute_reply":"2023-04-25T10:31:54.33506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scores statistics over targets ","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.hist(l)\nplt.show()\npd.Series(l).describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:31:54.338405Z","iopub.execute_input":"2023-04-25T10:31:54.338798Z","iopub.status.idle":"2023-04-25T10:31:54.565371Z","shell.execute_reply.started":"2023-04-25T10:31:54.338761Z","shell.execute_reply":"2023-04-25T10:31:54.564196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Retrain model on the the full sample ","metadata":{}},{"cell_type":"code","source":"%%time\nmodel.fit(X,Y)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:31:54.566885Z","iopub.execute_input":"2023-04-25T10:31:54.567259Z","iopub.status.idle":"2023-04-25T10:33:04.304246Z","shell.execute_reply.started":"2023-04-25T10:31:54.567224Z","shell.execute_reply":"2023-04-25T10:33:04.302847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission preparations Step 1 - load features and calculate predictions ","metadata":{}},{"cell_type":"markdown","source":"## Load features for submission","metadata":{}},{"cell_type":"code","source":"%%time\n# fn = '/kaggle/input/protein-embeddings-1/reduced_embeddings_file.npy'\n# fn = '/kaggle/input/protein-embeddings-1/embed_protbert_train_clip_1200_first_70000_prot.csv'\n# fn = '/kaggle/input/t5embeds/train_embeds.npy'\nfn = '/kaggle/input/t5embeds/test_embeds.npy'\nprint(fn)\nX_submit = np.load(fn)\nprint(X_submit.shape)\n# X_submit","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:33:04.306794Z","iopub.execute_input":"2023-04-25T10:33:04.307857Z","iopub.status.idle":"2023-04-25T10:33:15.505258Z","shell.execute_reply.started":"2023-04-25T10:33:04.307795Z","shell.execute_reply":"2023-04-25T10:33:15.503759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Calculate prediction for submission","metadata":{}},{"cell_type":"code","source":"%%time\nY_submit =  model.predict(X_submit)\nprint(Y_submit.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:33:15.506819Z","iopub.execute_input":"2023-04-25T10:33:15.507179Z","iopub.status.idle":"2023-04-25T10:33:22.900724Z","shell.execute_reply.started":"2023-04-25T10:33:15.507145Z","shell.execute_reply":"2023-04-25T10:33:22.89931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission preparations Step 2 - prepare submision in desired format  ","metadata":{}},{"cell_type":"code","source":"%%time \ndf_finalSubmission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:33:22.902424Z","iopub.execute_input":"2023-04-25T10:33:22.902819Z","iopub.status.idle":"2023-04-25T10:33:22.913267Z","shell.execute_reply.started":"2023-04-25T10:33:22.902781Z","shell.execute_reply":"2023-04-25T10:33:22.911629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load protein ids for the submission","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/test_ids.npy'\nvec_test_protein_ids = np.load(fn)\nprint(vec_test_protein_ids.shape)\nvec_test_protein_ids","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:33:22.915408Z","iopub.execute_input":"2023-04-25T10:33:22.915919Z","iopub.status.idle":"2023-04-25T10:33:22.97852Z","shell.execute_reply.started":"2023-04-25T10:33:22.915866Z","shell.execute_reply":"2023-04-25T10:33:22.977617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## \"Melt\" protein ids ","metadata":{}},{"cell_type":"code","source":"%%time \nl = []\nfor k in list(vec_test_protein_ids):\n    l += [ k] * Y_submit.shape[1]\nprint(len(l), l[:20])    \n\ndf_finalSubmission['Protein Id'] = l","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:33:22.980025Z","iopub.execute_input":"2023-04-25T10:33:22.980375Z","iopub.status.idle":"2023-04-25T10:34:25.580352Z","shell.execute_reply.started":"2023-04-25T10:33:22.980341Z","shell.execute_reply":"2023-04-25T10:34:25.579067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %%time \n# df_finalSubmission.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:34:25.582115Z","iopub.execute_input":"2023-04-25T10:34:25.582466Z","iopub.status.idle":"2023-04-25T10:34:25.586903Z","shell.execute_reply.started":"2023-04-25T10:34:25.582433Z","shell.execute_reply":"2023-04-25T10:34:25.585407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## \"Melt\" Labels (Gene ontology terms )","metadata":{"execution":{"iopub.status.busy":"2023-04-24T12:31:15.267721Z","iopub.execute_input":"2023-04-24T12:31:15.268181Z","iopub.status.idle":"2023-04-24T12:31:15.613742Z","shell.execute_reply.started":"2023-04-24T12:31:15.268143Z","shell.execute_reply":"2023-04-24T12:31:15.612514Z"}}},{"cell_type":"code","source":"df_finalSubmission['GO Term Id'] = labels_to_consider * Y_submit.shape[0]\n# df_finalSubmission.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:34:25.594372Z","iopub.execute_input":"2023-04-25T10:34:25.595407Z","iopub.status.idle":"2023-04-25T10:34:40.972938Z","shell.execute_reply.started":"2023-04-25T10:34:25.595362Z","shell.execute_reply":"2023-04-25T10:34:40.971382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Assign predictions ","metadata":{}},{"cell_type":"code","source":"df_finalSubmission['Prediction'] = Y_submit.ravel()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:34:40.974345Z","iopub.execute_input":"2023-04-25T10:34:40.974777Z","iopub.status.idle":"2023-04-25T10:34:50.377716Z","shell.execute_reply.started":"2023-04-25T10:34:40.974737Z","shell.execute_reply":"2023-04-25T10:34:50.376249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df_finalSubmission)","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:25:05.604224Z","iopub.execute_input":"2023-04-25T11:25:05.605242Z","iopub.status.idle":"2023-04-25T11:25:05.623247Z","shell.execute_reply.started":"2023-04-25T11:25:05.605184Z","shell.execute_reply":"2023-04-25T11:25:05.621823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### drop 0 preds and negatives\n* opt: sort by score, keep top K per Protein\n* warning : will be slooow with this many rows!","metadata":{}},{"cell_type":"code","source":"%%time\ndf_finalSubmission['Prediction'] = df_finalSubmission['Prediction'].round(3)\ndf_finalSubmission = df_finalSubmission[df_finalSubmission['Prediction']>0]\ndf_finalSubmission.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:27:53.563708Z","iopub.execute_input":"2023-04-25T11:27:53.564928Z","iopub.status.idle":"2023-04-25T11:28:07.494003Z","shell.execute_reply.started":"2023-04-25T11:27:53.564863Z","shell.execute_reply":"2023-04-25T11:28:07.49293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:29:12.551566Z","iopub.execute_input":"2023-04-25T11:29:12.552033Z","iopub.status.idle":"2023-04-25T11:29:12.569386Z","shell.execute_reply.started":"2023-04-25T11:29:12.551988Z","shell.execute_reply":"2023-04-25T11:29:12.567915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save ","metadata":{}},{"cell_type":"code","source":"%%time \ndf_finalSubmission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:34:50.379208Z","iopub.execute_input":"2023-04-25T10:34:50.379612Z","iopub.status.idle":"2023-04-25T10:44:32.843101Z","shell.execute_reply.started":"2023-04-25T10:34:50.37957Z","shell.execute_reply":"2023-04-25T10:44:32.841681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show some info ","metadata":{}},{"cell_type":"code","source":"# %%time \n# df_finalSubmission.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:44:32.844817Z","iopub.execute_input":"2023-04-25T10:44:32.845237Z","iopub.status.idle":"2023-04-25T10:44:32.866844Z","shell.execute_reply.started":"2023-04-25T10:44:32.845196Z","shell.execute_reply":"2023-04-25T10:44:32.865049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \ndf_finalSubmission.describe()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:44:32.868617Z","iopub.execute_input":"2023-04-25T10:44:32.869072Z","iopub.status.idle":"2023-04-25T10:44:47.348079Z","shell.execute_reply.started":"2023-04-25T10:44:32.869027Z","shell.execute_reply":"2023-04-25T10:44:47.346888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nplt.figure(figsize = (15,4))\nplt.hist(df_finalSubmission['Prediction'].values, bins = 300 )\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:24:11.695153Z","iopub.execute_input":"2023-04-25T11:24:11.695601Z","iopub.status.idle":"2023-04-25T11:24:15.935234Z","shell.execute_reply.started":"2023-04-25T11:24:11.695561Z","shell.execute_reply":"2023-04-25T11:24:15.933793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission","metadata":{"execution":{"iopub.status.busy":"2023-04-25T10:44:55.196904Z","iopub.execute_input":"2023-04-25T10:44:55.197568Z","iopub.status.idle":"2023-04-25T10:44:55.212339Z","shell.execute_reply.started":"2023-04-25T10:44:55.197511Z","shell.execute_reply":"2023-04-25T10:44:55.210841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission.iloc[:,0:2].nunique()","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:21:45.587814Z","iopub.execute_input":"2023-04-25T11:21:45.588221Z","iopub.status.idle":"2023-04-25T11:22:26.511486Z","shell.execute_reply.started":"2023-04-25T11:21:45.588185Z","shell.execute_reply":"2023-04-25T11:22:26.510099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission.shape[0]/141864 ## num proteins","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:23:13.270168Z","iopub.execute_input":"2023-04-25T11:23:13.270637Z","iopub.status.idle":"2023-04-25T11:23:13.279507Z","shell.execute_reply.started":"2023-04-25T11:23:13.270595Z","shell.execute_reply":"2023-04-25T11:23:13.278076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission.shape[0]/n_labels_to_consider","metadata":{"execution":{"iopub.status.busy":"2023-04-25T11:22:27.913927Z","iopub.execute_input":"2023-04-25T11:22:27.915055Z","iopub.status.idle":"2023-04-25T11:22:27.924431Z","shell.execute_reply.started":"2023-04-25T11:22:27.914995Z","shell.execute_reply":"2023-04-25T11:22:27.922877Z"},"trusted":true},"execution_count":null,"outputs":[]}]}