{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":41875,"databundleVersionId":5521661,"sourceType":"competition"},{"sourceId":5499219,"sourceType":"datasetVersion","datasetId":3167603}],"dockerImageVersionId":30497,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"n_labels_to_consider = 1499 # We will choose only top frequent labels (in train) and predict only them. \nn_max_preds = 1499","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-29T00:34:16.83088Z","iopub.execute_input":"2023-05-29T00:34:16.831257Z","iopub.status.idle":"2023-05-29T00:34:16.845201Z","shell.execute_reply.started":"2023-05-29T00:34:16.831216Z","shell.execute_reply":"2023-05-29T00:34:16.843091Z"},"trusted":true},"execution_count":1,"outputs":[]},{"cell_type":"code","source":"import time\nt0start = time.time() \n\nimport numpy as np\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom sklearn.model_selection import train_test_split, cross_val_score, GridSearchCV, KFold, RandomizedSearchCV\nfrom sklearn.linear_model import Ridge,RidgeCV\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.multioutput import MultiOutputClassifier\n\nimport xgboost as xgb\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2023-05-29T00:34:20.318616Z","iopub.execute_input":"2023-05-29T00:34:20.319508Z","iopub.status.idle":"2023-05-29T00:34:22.892231Z","shell.execute_reply.started":"2023-05-29T00:34:20.319462Z","shell.execute_reply":"2023-05-29T00:34:22.891255Z"},"trusted":true},"execution_count":2,"outputs":[{"name":"stdout","text":"/kaggle/input/t5embeds/train_ids.npy\n/kaggle/input/t5embeds/test_embeds.npy\n/kaggle/input/t5embeds/train_embeds.npy\n/kaggle/input/t5embeds/test_ids.npy\n/kaggle/input/cafa-5-protein-function-prediction/sample_submission.tsv\n/kaggle/input/cafa-5-protein-function-prediction/IA.txt\n/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta\n/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset-taxon-list.tsv\n/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\n/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\n/kaggle/input/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv\n/kaggle/input/cafa-5-protein-function-prediction/Train/go-basic.obo\n","output_type":"stream"}]},{"cell_type":"code","source":"%%time\ntrainTerms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\",sep=\"\\t\")\nprint(trainTerms.shape)\ndisplay(trainTerms.head(2))\nvec_freqCount = (trainTerms['term'].value_counts())\nprint(vec_freqCount )","metadata":{"execution":{"iopub.status.busy":"2023-05-29T00:34:26.479251Z","iopub.execute_input":"2023-05-29T00:34:26.479614Z","iopub.status.idle":"2023-05-29T00:34:32.477811Z","shell.execute_reply.started":"2023-05-29T00:34:26.479586Z","shell.execute_reply":"2023-05-29T00:34:32.476853Z"},"trusted":true},"execution_count":3,"outputs":[{"name":"stdout","text":"(5363863, 3)\n","output_type":"stream"},{"output_type":"display_data","data":{"text/plain":"      EntryID        term aspect\n0  A0A009IHW8  GO:0008152    BPO\n1  A0A009IHW8  GO:0034655    BPO","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>EntryID</th>\n      <th>term</th>\n      <th>aspect</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>A0A009IHW8</td>\n      <td>GO:0008152</td>\n      <td>BPO</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>A0A009IHW8</td>\n      <td>GO:0034655</td>\n      <td>BPO</td>\n    </tr>\n  </tbody>\n</table>\n</div>"},"metadata":{}},{"name":"stdout","text":"GO:0005575    92912\nGO:0008150    92210\nGO:0110165    91286\nGO:0003674    78637\nGO:0005622    70785\n              ...  \nGO:0031772        1\nGO:0042324        1\nGO:0031771        1\nGO:0051041        1\nGO:0102628        1\nName: term, Length: 31466, dtype: int64\nCPU times: user 2.8 s, sys: 329 ms, total: 3.13 s\nWall time: 5.99 s\n","output_type":"stream"}]},{"cell_type":"code","source":"## drop very rares\nvec_freqCount = vec_freqCount[vec_freqCount>=30]\nprint(vec_freqCount.shape[0])\nvec_freqCount.describe().round()","metadata":{"execution":{"iopub.status.busy":"2023-05-29T00:34:37.982355Z","iopub.execute_input":"2023-05-29T00:34:37.982706Z","iopub.status.idle":"2023-05-29T00:34:38.002289Z","shell.execute_reply.started":"2023-05-29T00:34:37.982678Z","shell.execute_reply":"2023-05-29T00:34:38.001225Z"},"trusted":true},"execution_count":4,"outputs":[{"name":"stdout","text":"8632\n","output_type":"stream"},{"execution_count":4,"output_type":"execute_result","data":{"text/plain":"count     8632.0\nmean       602.0\nstd       3255.0\nmin         30.0\n25%         49.0\n50%         93.0\n75%        261.0\nmax      92912.0\nName: term, dtype: float64"},"metadata":{}}]},{"cell_type":"code","source":"vec_freqCount[vec_freqCount>200].shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-05-29T00:34:41.990412Z","iopub.execute_input":"2023-05-29T00:34:41.990768Z","iopub.status.idle":"2023-05-29T00:34:41.998458Z","shell.execute_reply.started":"2023-05-29T00:34:41.990739Z","shell.execute_reply":"2023-05-29T00:34:41.997548Z"},"trusted":true},"execution_count":5,"outputs":[{"execution_count":5,"output_type":"execute_result","data":{"text/plain":"2597"},"metadata":{}}]},{"cell_type":"code","source":"labels_to_consider = list(vec_freqCount.index[:n_labels_to_consider] )\nprint('n_labels_to_consider:', len(labels_to_consider), 'First 10:', labels_to_consider[:10] ) ","metadata":{"execution":{"iopub.status.busy":"2023-05-29T00:34:46.784508Z","iopub.execute_input":"2023-05-29T00:34:46.784862Z","iopub.status.idle":"2023-05-29T00:34:46.791415Z","shell.execute_reply.started":"2023-05-29T00:34:46.784832Z","shell.execute_reply":"2023-05-29T00:34:46.790132Z"},"trusted":true},"execution_count":6,"outputs":[{"name":"stdout","text":"n_labels_to_consider: 1499 First 10: ['GO:0005575', 'GO:0008150', 'GO:0110165', 'GO:0003674', 'GO:0005622', 'GO:0009987', 'GO:0043226', 'GO:0043229', 'GO:0005488', 'GO:0043227']\n","output_type":"stream"}]},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/train_ids.npy'\nvec_train_protein_ids = np.load(fn)\nprint(vec_train_protein_ids.shape)\nvec_train_protein_ids","metadata":{"execution":{"iopub.status.busy":"2023-05-29T00:34:50.164792Z","iopub.execute_input":"2023-05-29T00:34:50.165498Z","iopub.status.idle":"2023-05-29T00:34:50.228315Z","shell.execute_reply.started":"2023-05-29T00:34:50.165463Z","shell.execute_reply":"2023-05-29T00:34:50.227228Z"},"trusted":true},"execution_count":7,"outputs":[{"name":"stdout","text":"(142246,)\nCPU times: user 1.36 ms, sys: 6.16 ms, total: 7.53 ms\nWall time: 53.7 ms\n","output_type":"stream"},{"execution_count":7,"output_type":"execute_result","data":{"text/plain":"array(['P20536', 'O73864', 'O95231', ..., 'Q5RGB0', 'A0A2R8QMZ5',\n       'A0A8I6GHU0'], dtype='<U10')"},"metadata":{}}]},{"cell_type":"code","source":"%%time \ntrain_size = 142246 # len(X)\nY = np.zeros( (train_size ,n_labels_to_consider) )\nprint(Y.shape)\n\nseries_train_protein_ids = pd.Series(vec_train_protein_ids ) # \n\ntrainTerms_smaller = trainTerms[ trainTerms['term'].isin( labels_to_consider ) ] # to speed-up the next step \nprint( trainTerms_smaller.shape)\n\nfor i in range(Y.shape[1]):\n    m = trainTerms_smaller['term'] ==  labels_to_consider[i]\n#     m.sum()\n    Y[:,i] =  series_train_protein_ids.isin(  set(trainTerms_smaller[m]['EntryID'] ) ).astype(float )\n    if (i % 10) == 0: \n        print(i, m.sum())\nY","metadata":{"execution":{"iopub.status.busy":"2023-05-29T00:34:58.410181Z","iopub.execute_input":"2023-05-29T00:34:58.410564Z"},"trusted":true},"execution_count":null,"outputs":[{"name":"stdout","text":"(142246, 1499)\n(4420307, 3)\n0 92912\n10 53193\n20 28680\n30 19458\n40 16657\n50 14757\n60 12963\n70 11013\n80 10236\n90 9350\n100 8475\n110 7664\n120 7099\n130 6522\n140 5866\n150 5316\n160 5084\n170 4831\n180 4585\n190 4383\n200 4214\n210 4091\n220 3788\n230 3652\n240 3535\n250 3358\n260 3231\n270 3115\n280 3047\n290 2952\n300 2867\n310 2806\n320 2764\n330 2674\n","output_type":"stream"}]},{"cell_type":"code","source":"%%time \n# save for possible future reuse \nfn4saveY = 'Y_'+str(Y.shape[1])\nprint(fn4saveY)\nnp.save( fn4saveY , Y) ","metadata":{"execution":{"iopub.status.busy":"2023-05-27T17:39:44.967143Z","iopub.execute_input":"2023-05-27T17:39:44.967519Z","iopub.status.idle":"2023-05-27T17:39:46.342131Z","shell.execute_reply.started":"2023-05-27T17:39:44.967483Z","shell.execute_reply":"2023-05-27T17:39:46.340668Z"},"trusted":true},"execution_count":12,"outputs":[{"name":"stdout","text":"Y_1499\nCPU times: user 3.78 ms, sys: 1.36 s, total: 1.36 s\nWall time: 1.36 s\n","output_type":"stream"}]},{"cell_type":"code","source":"%%time\nfn4save_labels = 'Y_'+str(Y.shape[1]) + '_labels'\nnp.save(fn4save_labels, labels_to_consider )","metadata":{"execution":{"iopub.status.busy":"2023-05-27T17:39:46.344458Z","iopub.execute_input":"2023-05-27T17:39:46.345394Z","iopub.status.idle":"2023-05-27T17:39:46.363305Z","shell.execute_reply.started":"2023-05-27T17:39:46.345328Z","shell.execute_reply":"2023-05-27T17:39:46.361963Z"},"trusted":true},"execution_count":13,"outputs":[{"name":"stdout","text":"CPU times: user 1.93 ms, sys: 0 ns, total: 1.93 ms\nWall time: 4.01 ms\n","output_type":"stream"}]},{"cell_type":"code","source":"%%time \n# Someone may prefer  Y as dataframe \nif 1:\n    df_Y = pd.DataFrame(data = Y, columns = labels_to_consider)\n    display(df_Y.head(2))\n#     print( df.info().sum() )\n    print('memory_usage:', df_Y.memory_usage(index=True).sum() )\n    display(df_Y.describe() )    \n    fn4save =  'df_Y_'+str(Y.shape[1]) + '.csv'\n    df_Y.to_csv(fn4save)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T17:39:46.367982Z","iopub.execute_input":"2023-05-27T17:39:46.368862Z","iopub.status.idle":"2023-05-27T17:42:08.711252Z","shell.execute_reply.started":"2023-05-27T17:39:46.368826Z","shell.execute_reply":"2023-05-27T17:42:08.710289Z"},"trusted":true},"execution_count":14,"outputs":[{"output_type":"display_data","data":{"text/plain":"   GO:0005575  GO:0008150  GO:0110165  GO:0003674  GO:0005622  GO:0009987  \\\n0         0.0         1.0         0.0         1.0         0.0         1.0   \n1         1.0         1.0         1.0         1.0         0.0         1.0   \n\n   GO:0043226  GO:0043229  GO:0005488  GO:0043227  ...  GO:0000313  \\\n0         0.0         0.0         1.0         0.0  ...         0.0   \n1         0.0         0.0         1.0         0.0  ...         0.0   \n\n   GO:0034250  GO:0140053  GO:0031345  GO:0098802  GO:0045861  GO:0051783  \\\n0         0.0         0.0         0.0         0.0         0.0         0.0   \n1         0.0         0.0         0.0         0.0         0.0         0.0   \n\n   GO:0031674  GO:0001818  GO:0006874  \n0         0.0         0.0         0.0  \n1         0.0         0.0         0.0  \n\n[2 rows x 1499 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>GO:0005575</th>\n      <th>GO:0008150</th>\n      <th>GO:0110165</th>\n      <th>GO:0003674</th>\n      <th>GO:0005622</th>\n      <th>GO:0009987</th>\n      <th>GO:0043226</th>\n      <th>GO:0043229</th>\n      <th>GO:0005488</th>\n      <th>GO:0043227</th>\n      <th>...</th>\n      <th>GO:0000313</th>\n      <th>GO:0034250</th>\n      <th>GO:0140053</th>\n      <th>GO:0031345</th>\n      <th>GO:0098802</th>\n      <th>GO:0045861</th>\n      <th>GO:0051783</th>\n      <th>GO:0031674</th>\n      <th>GO:0001818</th>\n      <th>GO:0006874</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>0.0</td>\n      <td>1.0</td>\n      <td>0.0</td>\n      <td>1.0</td>\n      <td>0.0</td>\n      <td>1.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>1.0</td>\n      <td>0.0</td>\n      <td>...</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>1.0</td>\n      <td>1.0</td>\n      <td>1.0</td>\n      <td>1.0</td>\n      <td>0.0</td>\n      <td>1.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>1.0</td>\n      <td>0.0</td>\n      <td>...</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n      <td>0.0</td>\n    </tr>\n  </tbody>\n</table>\n<p>2 rows × 1499 columns</p>\n</div>"},"metadata":{}},{"name":"stdout","text":"memory_usage: 1705814160\n","output_type":"stream"},{"output_type":"display_data","data":{"text/plain":"          GO:0005575     GO:0008150     GO:0110165     GO:0003674  \\\ncount  142246.000000  142246.000000  142246.000000  142246.000000   \nmean        0.653178       0.648243       0.641747       0.552824   \nstd         0.475960       0.477520       0.479489       0.497204   \nmin         0.000000       0.000000       0.000000       0.000000   \n25%         0.000000       0.000000       0.000000       0.000000   \n50%         1.000000       1.000000       1.000000       1.000000   \n75%         1.000000       1.000000       1.000000       1.000000   \nmax         1.000000       1.000000       1.000000       1.000000   \n\n          GO:0005622     GO:0009987     GO:0043226     GO:0043229  \\\ncount  142246.000000  142246.000000  142246.000000  142246.000000   \nmean        0.497624       0.430894       0.428012       0.409959   \nstd         0.499996       0.495203       0.494792       0.491827   \nmin         0.000000       0.000000       0.000000       0.000000   \n25%         0.000000       0.000000       0.000000       0.000000   \n50%         0.000000       0.000000       0.000000       0.000000   \n75%         1.000000       1.000000       1.000000       1.000000   \nmax         1.000000       1.000000       1.000000       1.000000   \n\n          GO:0005488     GO:0043227  ...     GO:0000313     GO:0034250  \\\ncount  142246.000000  142246.000000  ...  142246.000000  142246.000000   \nmean        0.403386       0.389832  ...       0.003058       0.003051   \nstd         0.490579       0.487714  ...       0.055216       0.055152   \nmin         0.000000       0.000000  ...       0.000000       0.000000   \n25%         0.000000       0.000000  ...       0.000000       0.000000   \n50%         0.000000       0.000000  ...       0.000000       0.000000   \n75%         1.000000       1.000000  ...       0.000000       0.000000   \nmax         1.000000       1.000000  ...       1.000000       1.000000   \n\n          GO:0140053     GO:0031345     GO:0098802     GO:0045861  \\\ncount  142246.000000  142246.000000  142246.000000  142246.000000   \nmean        0.003051       0.003051       0.003044       0.003037   \nstd         0.055152       0.055152       0.055089       0.055025   \nmin         0.000000       0.000000       0.000000       0.000000   \n25%         0.000000       0.000000       0.000000       0.000000   \n50%         0.000000       0.000000       0.000000       0.000000   \n75%         0.000000       0.000000       0.000000       0.000000   \nmax         1.000000       1.000000       1.000000       1.000000   \n\n          GO:0051783     GO:0031674     GO:0001818     GO:0006874  \ncount  142246.000000  142246.000000  142246.000000  142246.000000  \nmean        0.003030       0.003030       0.003030       0.003023  \nstd         0.054962       0.054962       0.054962       0.054898  \nmin         0.000000       0.000000       0.000000       0.000000  \n25%         0.000000       0.000000       0.000000       0.000000  \n50%         0.000000       0.000000       0.000000       0.000000  \n75%         0.000000       0.000000       0.000000       0.000000  \nmax         1.000000       1.000000       1.000000       1.000000  \n\n[8 rows x 1499 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>GO:0005575</th>\n      <th>GO:0008150</th>\n      <th>GO:0110165</th>\n      <th>GO:0003674</th>\n      <th>GO:0005622</th>\n      <th>GO:0009987</th>\n      <th>GO:0043226</th>\n      <th>GO:0043229</th>\n      <th>GO:0005488</th>\n      <th>GO:0043227</th>\n      <th>...</th>\n      <th>GO:0000313</th>\n      <th>GO:0034250</th>\n      <th>GO:0140053</th>\n      <th>GO:0031345</th>\n      <th>GO:0098802</th>\n      <th>GO:0045861</th>\n      <th>GO:0051783</th>\n      <th>GO:0031674</th>\n      <th>GO:0001818</th>\n      <th>GO:0006874</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>count</th>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>...</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n      <td>142246.000000</td>\n    </tr>\n    <tr>\n      <th>mean</th>\n      <td>0.653178</td>\n      <td>0.648243</td>\n      <td>0.641747</td>\n      <td>0.552824</td>\n      <td>0.497624</td>\n      <td>0.430894</td>\n      <td>0.428012</td>\n      <td>0.409959</td>\n      <td>0.403386</td>\n      <td>0.389832</td>\n      <td>...</td>\n      <td>0.003058</td>\n      <td>0.003051</td>\n      <td>0.003051</td>\n      <td>0.003051</td>\n      <td>0.003044</td>\n      <td>0.003037</td>\n      <td>0.003030</td>\n      <td>0.003030</td>\n      <td>0.003030</td>\n      <td>0.003023</td>\n    </tr>\n    <tr>\n      <th>std</th>\n      <td>0.475960</td>\n      <td>0.477520</td>\n      <td>0.479489</td>\n      <td>0.497204</td>\n      <td>0.499996</td>\n      <td>0.495203</td>\n      <td>0.494792</td>\n      <td>0.491827</td>\n      <td>0.490579</td>\n      <td>0.487714</td>\n      <td>...</td>\n      <td>0.055216</td>\n      <td>0.055152</td>\n      <td>0.055152</td>\n      <td>0.055152</td>\n      <td>0.055089</td>\n      <td>0.055025</td>\n      <td>0.054962</td>\n      <td>0.054962</td>\n      <td>0.054962</td>\n      <td>0.054898</td>\n    </tr>\n    <tr>\n      <th>min</th>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>...</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n    </tr>\n    <tr>\n      <th>25%</th>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>...</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n    </tr>\n    <tr>\n      <th>50%</th>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>...</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n    </tr>\n    <tr>\n      <th>75%</th>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>...</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n      <td>0.000000</td>\n    </tr>\n    <tr>\n      <th>max</th>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>...</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n      <td>1.000000</td>\n    </tr>\n  </tbody>\n</table>\n<p>8 rows × 1499 columns</p>\n</div>"},"metadata":{}},{"name":"stdout","text":"CPU times: user 2min 19s, sys: 1.74 s, total: 2min 21s\nWall time: 2min 22s\n","output_type":"stream"}]},{"cell_type":"code","source":"%%time\n\n# fn = '/kaggle/input/protein-embeddings-1/reduced_embeddings_file.npy'\n# fn = '/kaggle/input/protein-embeddings-1/embed_protbert_train_clip_1200_first_70000_prot.csv'\nfn = '/kaggle/input/t5embeds/train_embeds.npy'\n# fn = '/kaggle/input/t5embeds/test_embeds.npy'\n\nprint(fn)\nif '.csv' in fn:\n    df = pd.read_csv(fn, index_col = 0)\n    X = df.values\nelif '.npy' in fn:\n    X = np.load(fn)\nprint(X.shape)\nX","metadata":{"execution":{"iopub.status.busy":"2023-05-28T23:56:46.429168Z","iopub.execute_input":"2023-05-28T23:56:46.430293Z","iopub.status.idle":"2023-05-28T23:56:56.131347Z","shell.execute_reply.started":"2023-05-28T23:56:46.430245Z","shell.execute_reply":"2023-05-28T23:56:56.130305Z"},"trusted":true},"execution_count":9,"outputs":[{"name":"stdout","text":"/kaggle/input/t5embeds/train_embeds.npy\n(142246, 1024)\nCPU times: user 0 ns, sys: 683 ms, total: 683 ms\nWall time: 9.69 s\n","output_type":"stream"},{"execution_count":9,"output_type":"execute_result","data":{"text/plain":"array([[ 0.04948843, -0.03293516,  0.03247323, ..., -0.04353154,\n         0.0964628 ,  0.07306959],\n       [-0.04461636,  0.06492499, -0.08026284, ...,  0.02672353,\n         0.02787905, -0.04842958],\n       [-0.02012804, -0.04977943,  0.00789446, ..., -0.03610279,\n         0.00769301,  0.10623412],\n       ...,\n       [ 0.01691809,  0.04133058,  0.00079253, ...,  0.0088079 ,\n         0.00648063, -0.01334958],\n       [ 0.06125151,  0.08340203,  0.0440247 , ...,  0.00138361,\n        -0.04754627,  0.01012351],\n       [ 0.02160021,  0.06516985,  0.07492343, ...,  0.0496657 ,\n        -0.01987522,  0.04471432]])"},"metadata":{}}]},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/train_ids.npy'\nvec_train_protein_ids = np.load(fn)\nprint(vec_train_protein_ids.shape)\nvec_train_protein_ids","metadata":{"execution":{"iopub.status.busy":"2023-05-28T23:56:59.281551Z","iopub.execute_input":"2023-05-28T23:56:59.281936Z","iopub.status.idle":"2023-05-28T23:56:59.295484Z","shell.execute_reply.started":"2023-05-28T23:56:59.281906Z","shell.execute_reply":"2023-05-28T23:56:59.293766Z"},"trusted":true},"execution_count":10,"outputs":[{"name":"stdout","text":"(142246,)\nCPU times: user 2.55 ms, sys: 3 ms, total: 5.55 ms\nWall time: 5.31 ms\n","output_type":"stream"},{"execution_count":10,"output_type":"execute_result","data":{"text/plain":"array(['P20536', 'O73864', 'O95231', ..., 'Q5RGB0', 'A0A2R8QMZ5',\n       'A0A8I6GHU0'], dtype='<U10')"},"metadata":{}}]},{"cell_type":"code","source":"print(X.shape)\nprint(len(X))","metadata":{"execution":{"iopub.status.busy":"2023-05-28T23:57:00.757281Z","iopub.execute_input":"2023-05-28T23:57:00.757883Z","iopub.status.idle":"2023-05-28T23:57:00.763125Z","shell.execute_reply.started":"2023-05-28T23:57:00.757848Z","shell.execute_reply":"2023-05-28T23:57:00.762026Z"},"trusted":true},"execution_count":11,"outputs":[{"name":"stdout","text":"(142246, 1024)\n142246\n","output_type":"stream"}]},{"cell_type":"code","source":"IX = np.arange(len(X))\nprint(IX.shape)\nprint(IX)\nIX_train, IX_test, _,_ = train_test_split( IX, IX, train_size=0.1, random_state=42)\n# print(len(IX_train), len(IX_test),  IX_train[:10], IX_test[:10] )","metadata":{"execution":{"iopub.status.busy":"2023-05-28T23:57:02.562801Z","iopub.execute_input":"2023-05-28T23:57:02.563153Z","iopub.status.idle":"2023-05-28T23:57:02.578326Z","shell.execute_reply.started":"2023-05-28T23:57:02.563125Z","shell.execute_reply":"2023-05-28T23:57:02.576732Z"},"trusted":true},"execution_count":12,"outputs":[{"name":"stdout","text":"(142246,)\n[     0      1      2 ... 142243 142244 142245]\n","output_type":"stream"}]},{"cell_type":"code","source":"clf_xgb = xgb.XGBClassifier(objective=\"binary:logistic\", random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T23:57:04.143782Z","iopub.execute_input":"2023-05-28T23:57:04.144452Z","iopub.status.idle":"2023-05-28T23:57:04.150654Z","shell.execute_reply.started":"2023-05-28T23:57:04.144421Z","shell.execute_reply":"2023-05-28T23:57:04.149452Z"},"trusted":true},"execution_count":13,"outputs":[]},{"cell_type":"code","source":"multilabel_xgb = MultiOutputClassifier(clf_xgb)","metadata":{"execution":{"iopub.status.busy":"2023-05-28T23:57:05.403521Z","iopub.execute_input":"2023-05-28T23:57:05.40386Z","iopub.status.idle":"2023-05-28T23:57:05.408264Z","shell.execute_reply.started":"2023-05-28T23:57:05.403834Z","shell.execute_reply":"2023-05-28T23:57:05.407078Z"},"trusted":true},"execution_count":14,"outputs":[]},{"cell_type":"code","source":"multilabel_xgb.fit(X[IX_train,:], Y[IX_train,:])","metadata":{"execution":{"iopub.status.busy":"2023-05-29T00:22:38.350423Z","iopub.execute_input":"2023-05-29T00:22:38.350769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_test = multilabel_xgb.predict(X[IX_train[:10],:])","metadata":{"execution":{"iopub.status.busy":"2023-05-29T00:19:54.824168Z","iopub.execute_input":"2023-05-29T00:19:54.825179Z","iopub.status.idle":"2023-05-29T00:19:56.599376Z","shell.execute_reply.started":"2023-05-29T00:19:54.825144Z","shell.execute_reply":"2023-05-29T00:19:56.598559Z"},"trusted":true},"execution_count":23,"outputs":[]},{"cell_type":"code","source":"print(y_pred_test)","metadata":{"execution":{"iopub.status.busy":"2023-05-29T00:20:06.806293Z","iopub.execute_input":"2023-05-29T00:20:06.806637Z","iopub.status.idle":"2023-05-29T00:20:06.814777Z","shell.execute_reply.started":"2023-05-29T00:20:06.806611Z","shell.execute_reply":"2023-05-29T00:20:06.813791Z"},"trusted":true},"execution_count":24,"outputs":[{"name":"stdout","text":"[[1 1 1 ... 0 0 0]\n [0 1 0 ... 0 0 0]\n [1 1 1 ... 0 0 0]\n ...\n [0 1 0 ... 0 0 0]\n [1 1 1 ... 0 0 0]\n [1 1 1 ... 0 0 0]]\n","output_type":"stream"}]}]}