{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Convert numpy predictions to CAFA submission format","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport xgboost as xgb\nfrom sklearn.metrics import confusion_matrix, roc_auc_score, accuracy_score, f1_score, roc_curve\nfrom sklearn.experimental import enable_halving_search_cv\nfrom sklearn.model_selection import train_test_split, GridSearchCV, HalvingGridSearchCV\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nfrom copy import deepcopy\nimport time\nfrom sklearn.multioutput import MultiOutputClassifier\n\nmpl.rcParams['figure.dpi'] = 200","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-14T05:00:32.905128Z","iopub.execute_input":"2023-07-14T05:00:32.905475Z","iopub.status.idle":"2023-07-14T05:00:32.912386Z","shell.execute_reply.started":"2023-07-14T05:00:32.905446Z","shell.execute_reply":"2023-07-14T05:00:32.911348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-07-14T05:00:45.258183Z","iopub.execute_input":"2023-07-14T05:00:45.258636Z","iopub.status.idle":"2023-07-14T05:00:45.264721Z","shell.execute_reply.started":"2023-07-14T05:00:45.258603Z","shell.execute_reply":"2023-07-14T05:00:45.263046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load GO labels","metadata":{}},{"cell_type":"code","source":"# labels_to_consider = np.load(\"/kaggle/input/xgbdata/Y_1499_labels.npy\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_to_consider = np.load('/kaggle/input/lstm-dataset/label-names-top-500.npy',allow_pickle=True)","metadata":{"execution":{"iopub.status.busy":"2023-07-14T05:01:05.855974Z","iopub.execute_input":"2023-07-14T05:01:05.856435Z","iopub.status.idle":"2023-07-14T05:01:05.865941Z","shell.execute_reply.started":"2023-07-14T05:01:05.856401Z","shell.execute_reply":"2023-07-14T05:01:05.865074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load all protein ids","metadata":{}},{"cell_type":"code","source":"fn = '/kaggle/input/protbert-embeddings-for-cafa5/test_ids.npy'\nvec_test_protein_ids = np.load(fn)","metadata":{"execution":{"iopub.status.busy":"2023-07-14T05:01:10.652149Z","iopub.execute_input":"2023-07-14T05:01:10.652535Z","iopub.status.idle":"2023-07-14T05:01:10.71895Z","shell.execute_reply.started":"2023-07-14T05:01:10.652504Z","shell.execute_reply":"2023-07-14T05:01:10.717979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load predections","metadata":{}},{"cell_type":"code","source":"Y_pred = np.load('/kaggle/input/bilstm-sub1/pred.npy')","metadata":{"execution":{"iopub.status.busy":"2023-07-14T05:01:14.181528Z","iopub.execute_input":"2023-07-14T05:01:14.18192Z","iopub.status.idle":"2023-07-14T05:01:18.030071Z","shell.execute_reply.started":"2023-07-14T05:01:14.181891Z","shell.execute_reply":"2023-07-14T05:01:18.028998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred","metadata":{"execution":{"iopub.status.busy":"2023-07-14T05:01:19.867929Z","iopub.execute_input":"2023-07-14T05:01:19.868375Z","iopub.status.idle":"2023-07-14T05:01:19.880381Z","shell.execute_reply.started":"2023-07-14T05:01:19.86834Z","shell.execute_reply":"2023-07-14T05:01:19.879058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-07-14T05:01:57.97157Z","iopub.execute_input":"2023-07-14T05:01:57.971958Z","iopub.status.idle":"2023-07-14T05:01:57.979964Z","shell.execute_reply.started":"2023-07-14T05:01:57.971929Z","shell.execute_reply":"2023-07-14T05:01:57.978557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_batch_sz = Y_pred.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-07-14T05:02:02.85871Z","iopub.execute_input":"2023-07-14T05:02:02.859097Z","iopub.status.idle":"2023-07-14T05:02:02.864194Z","shell.execute_reply.started":"2023-07-14T05:02:02.859069Z","shell.execute_reply":"2023-07-14T05:02:02.862885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_batch_sz","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])","metadata":{"execution":{"iopub.status.busy":"2023-07-14T05:02:05.652261Z","iopub.execute_input":"2023-07-14T05:02:05.652665Z","iopub.status.idle":"2023-07-14T05:02:05.660227Z","shell.execute_reply.started":"2023-07-14T05:02:05.652635Z","shell.execute_reply":"2023-07-14T05:02:05.659337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l = []\nfor k in list(vec_test_protein_ids[:test_data_batch_sz]):\n        l += [ k] * (Y_pred.shape[1])\nprint(len(l), l[:20])    \n\ndf_finalSubmission['Protein Id'] = l","metadata":{"execution":{"iopub.status.busy":"2023-07-14T05:02:08.450438Z","iopub.execute_input":"2023-07-14T05:02:08.450858Z","iopub.status.idle":"2023-07-14T05:02:36.854275Z","shell.execute_reply.started":"2023-07-14T05:02:08.450824Z","shell.execute_reply":"2023-07-14T05:02:36.853298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission['GO Term Id'] = [item for _ in range(test_data_batch_sz) for item in labels_to_consider]","metadata":{"execution":{"iopub.status.busy":"2023-07-14T05:02:50.739035Z","iopub.execute_input":"2023-07-14T05:02:50.739454Z","iopub.status.idle":"2023-07-14T05:03:00.536867Z","shell.execute_reply.started":"2023-07-14T05:02:50.739421Z","shell.execute_reply":"2023-07-14T05:03:00.535695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission['Prediction'] = Y_pred[:test_data_batch_sz].ravel()","metadata":{"execution":{"iopub.status.busy":"2023-07-14T05:03:06.595396Z","iopub.execute_input":"2023-07-14T05:03:06.595818Z","iopub.status.idle":"2023-07-14T05:03:06.972738Z","shell.execute_reply.started":"2023-07-14T05:03:06.595785Z","shell.execute_reply":"2023-07-14T05:03:06.971403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission.to_csv(\"submission.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"execution":{"iopub.status.busy":"2023-07-14T05:03:14.140471Z","iopub.execute_input":"2023-07-14T05:03:14.140896Z","iopub.status.idle":"2023-07-14T05:08:35.525885Z","shell.execute_reply.started":"2023-07-14T05:03:14.140862Z","shell.execute_reply":"2023-07-14T05:08:35.524766Z"},"trusted":true},"execution_count":null,"outputs":[]}]}