{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-16T19:51:49.481741Z","iopub.execute_input":"2023-05-16T19:51:49.48267Z","iopub.status.idle":"2023-05-16T19:51:49.529541Z","shell.execute_reply.started":"2023-05-16T19:51:49.482623Z","shell.execute_reply":"2023-05-16T19:51:49.528328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data loading","metadata":{}},{"cell_type":"code","source":"!pip install mygene scanpy","metadata":{"execution":{"iopub.status.busy":"2023-05-16T19:51:49.531887Z","iopub.execute_input":"2023-05-16T19:51:49.532231Z","iopub.status.idle":"2023-05-16T19:52:09.228419Z","shell.execute_reply.started":"2023-05-16T19:51:49.532197Z","shell.execute_reply":"2023-05-16T19:52:09.226866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from Bio import SeqIO\nimport pandas as pd\nimport mygene\nimport scanpy as sc\nimport anndata\nimport h5py\nfrom sklearn.decomposition import PCA\nfrom tqdm import tqdm\n\n\nmg = mygene.MyGeneInfo()\n\ntax_table = pd.read_table(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv\", sep=\"\\t\", index_col=\"EntryID\")\n    \nprot_names = [record.id for record in SeqIO.parse(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\", \"fasta\")]\n","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:00:11.108486Z","iopub.execute_input":"2023-05-16T21:00:11.10944Z","iopub.status.idle":"2023-05-16T21:00:14.141002Z","shell.execute_reply.started":"2023-05-16T21:00:11.109384Z","shell.execute_reply":"2023-05-16T21:00:14.139958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read Human scRNA-Seq and normalize","metadata":{}},{"cell_type":"code","source":"sc_hum_data = sc.read_h5ad(\"/kaggle/input/msci-raw-counts-anndata/train_cite_inputs_raw.ann.h5\")\nsc_hum_data.var_names = [x.split(\"_\")[0] for x in sc_hum_data.var_names]","metadata":{"execution":{"iopub.status.busy":"2023-05-16T19:52:15.590023Z","iopub.execute_input":"2023-05-16T19:52:15.591204Z","iopub.status.idle":"2023-05-16T19:52:44.578674Z","shell.execute_reply.started":"2023-05-16T19:52:15.591155Z","shell.execute_reply":"2023-05-16T19:52:44.577569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.pp.normalize_total(sc_hum_data, inplace=True)\nsc.pp.log1p(sc_hum_data)\nsc.pp.normalize_total(sc_hum_data, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T19:53:09.523191Z","iopub.execute_input":"2023-05-16T19:53:09.523775Z","iopub.status.idle":"2023-05-16T19:53:18.543412Z","shell.execute_reply.started":"2023-05-16T19:53:09.52371Z","shell.execute_reply":"2023-05-16T19:53:18.542001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read Mouse scRNA-Seq","metadata":{}},{"cell_type":"code","source":"sc_mus_data = sc.read_text(\"/kaggle/input/single-cell-rna-seq-nestorova2016-mouse-hspc/nestorowa_corrected_log2_transformed_counts.txt\")","metadata":{"execution":{"iopub.status.busy":"2023-05-16T19:53:18.545299Z","iopub.execute_input":"2023-05-16T19:53:18.546251Z","iopub.status.idle":"2023-05-16T19:53:20.582697Z","shell.execute_reply.started":"2023-05-16T19:53:18.546212Z","shell.execute_reply":"2023-05-16T19:53:20.581787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Select human and mouse proteins","metadata":{}},{"cell_type":"code","source":"hum_prot = tax_table.loc[tax_table[\"taxonomyID\"] == 9606].index.to_list()\nmus_prot = tax_table.loc[tax_table[\"taxonomyID\"] == 10090].index.to_list()","metadata":{"execution":{"iopub.status.busy":"2023-05-16T19:53:20.586679Z","iopub.execute_input":"2023-05-16T19:53:20.587112Z","iopub.status.idle":"2023-05-16T19:53:20.602683Z","shell.execute_reply.started":"2023-05-16T19:53:20.587077Z","shell.execute_reply":"2023-05-16T19:53:20.601738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Query MyGene to get gene names","metadata":{}},{"cell_type":"code","source":"hum_gene_prot_conv = mg.querymany(hum_prot, scopes='uniprot', fields='symbol,ensembl.gene', species='human', returnall=False, as_dataframe=True)\nmus_gene_prot_conv = mg.querymany(mus_prot, scopes='uniprot', fields='symbol,ensembl.gene', species='mouse', returnall=False, as_dataframe=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T19:53:20.604125Z","iopub.execute_input":"2023-05-16T19:53:20.604444Z","iopub.status.idle":"2023-05-16T19:54:55.759185Z","shell.execute_reply.started":"2023-05-16T19:53:20.604411Z","shell.execute_reply":"2023-05-16T19:54:55.758185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hum_gene_prot_conv","metadata":{"execution":{"iopub.status.busy":"2023-05-16T19:54:55.760467Z","iopub.execute_input":"2023-05-16T19:54:55.761269Z","iopub.status.idle":"2023-05-16T19:54:55.791969Z","shell.execute_reply.started":"2023-05-16T19:54:55.761234Z","shell.execute_reply":"2023-05-16T19:54:55.790733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mus_gene_prot_conv","metadata":{"execution":{"iopub.status.busy":"2023-05-16T19:54:55.793571Z","iopub.execute_input":"2023-05-16T19:54:55.794781Z","iopub.status.idle":"2023-05-16T19:54:55.812001Z","shell.execute_reply.started":"2023-05-16T19:54:55.794714Z","shell.execute_reply":"2023-05-16T19:54:55.810829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Subset scRNA-Seq to CAFA5 genes","metadata":{}},{"cell_type":"code","source":"hum_shared_genes = list(set(sc_hum_data.var_names).intersection(set(hum_gene_prot_conv[\"ensembl.gene\"].unique())))\ncafa_hum_gene_features = sc_hum_data[:, hum_shared_genes]","metadata":{"execution":{"iopub.status.busy":"2023-05-16T19:54:55.813553Z","iopub.execute_input":"2023-05-16T19:54:55.814238Z","iopub.status.idle":"2023-05-16T19:54:55.847067Z","shell.execute_reply.started":"2023-05-16T19:54:55.814197Z","shell.execute_reply":"2023-05-16T19:54:55.846151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cafa_hum_gene_features","metadata":{"execution":{"iopub.status.busy":"2023-05-16T19:54:55.848569Z","iopub.execute_input":"2023-05-16T19:54:55.849106Z","iopub.status.idle":"2023-05-16T19:54:55.855964Z","shell.execute_reply.started":"2023-05-16T19:54:55.849063Z","shell.execute_reply":"2023-05-16T19:54:55.854861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mus_shared_genes = list(set(sc_mus_data.var_names).intersection(set(mus_gene_prot_conv[\"symbol\"].unique())))\ncafa_mus_gene_features = sc_mus_data[:, mus_shared_genes]","metadata":{"execution":{"iopub.status.busy":"2023-05-16T19:54:55.85972Z","iopub.execute_input":"2023-05-16T19:54:55.860009Z","iopub.status.idle":"2023-05-16T19:54:55.872874Z","shell.execute_reply.started":"2023-05-16T19:54:55.859982Z","shell.execute_reply":"2023-05-16T19:54:55.872041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cafa_mus_gene_features","metadata":{"execution":{"iopub.status.busy":"2023-05-16T19:54:55.874066Z","iopub.execute_input":"2023-05-16T19:54:55.87433Z","iopub.status.idle":"2023-05-16T19:54:55.880877Z","shell.execute_reply.started":"2023-05-16T19:54:55.874303Z","shell.execute_reply":"2023-05-16T19:54:55.879893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PCA of scRNA-Seq to get new features","metadata":{}},{"cell_type":"code","source":"hum_gene_pca = PCA(n_components=100).fit_transform(cafa_hum_gene_features.X.T.todense())\nhum_gene_pca = pd.DataFrame(hum_gene_pca.T, columns = cafa_hum_gene_features.var_names.to_list())","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:19:30.023512Z","iopub.execute_input":"2023-05-16T21:19:30.024006Z","iopub.status.idle":"2023-05-16T21:20:26.778474Z","shell.execute_reply.started":"2023-05-16T21:19:30.023963Z","shell.execute_reply":"2023-05-16T21:20:26.777061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_feature_hum = []\nhum_proteins = []\nfor prot in tqdm(hum_prot):\n    try:\n        genes = hum_gene_prot_conv.loc[prot].dropna()[\"ensembl.gene\"]\n        if type(genes) == str:\n            genes = [genes]\n        for gene in genes:\n            if gene in hum_gene_pca.columns.to_list():\n                hum_proteins.append(prot)\n                new_feature_hum.append(hum_gene_pca[gene].to_list())\n    except:\n        continue\n        \nnew_features_hum = pd.DataFrame(new_feature_hum).T\nnew_features_hum.columns = hum_proteins\nnew_features_hum.to_csv(\"/kaggle/working/new_features_hum_genes.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:20:26.781005Z","iopub.execute_input":"2023-05-16T21:20:26.781381Z","iopub.status.idle":"2023-05-16T21:21:39.031551Z","shell.execute_reply.started":"2023-05-16T21:20:26.781344Z","shell.execute_reply":"2023-05-16T21:21:39.030049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_features_hum","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:21:39.033519Z","iopub.execute_input":"2023-05-16T21:21:39.033947Z","iopub.status.idle":"2023-05-16T21:21:39.078587Z","shell.execute_reply.started":"2023-05-16T21:21:39.033902Z","shell.execute_reply":"2023-05-16T21:21:39.076877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mus_gene_pca = PCA(n_components=100).fit_transform(cafa_mus_gene_features.X.T)\nmus_gene_pca = pd.DataFrame(mus_gene_pca.T, columns = cafa_mus_gene_features.var_names.to_list())","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:21:39.083518Z","iopub.execute_input":"2023-05-16T21:21:39.083997Z","iopub.status.idle":"2023-05-16T21:21:39.397785Z","shell.execute_reply.started":"2023-05-16T21:21:39.083953Z","shell.execute_reply":"2023-05-16T21:21:39.395869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_feature_mus = []\nmus_proteins = []\nfor prot in mus_prot:\n    try:\n        genes = mus_gene_prot_conv.loc[prot].dropna()[\"symbol\"]\n        if type(genes) == str:\n            genes = [genes]\n        for gene in genes:\n            if gene in mus_gene_pca.columns.to_list():\n                mus_proteins.append(prot)\n                new_feature_mus.append(mus_gene_pca[gene].to_list())\n    except:\n        continue\n        \nnew_features_mus = pd.DataFrame(new_feature_mus).T\nnew_features_mus.columns = mus_proteins\nnew_features_mus.to_csv(\"/kaggle/working/new_features_mus_genes.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:25:02.184554Z","iopub.execute_input":"2023-05-16T21:25:02.185002Z","iopub.status.idle":"2023-05-16T21:25:23.993294Z","shell.execute_reply.started":"2023-05-16T21:25:02.184957Z","shell.execute_reply":"2023-05-16T21:25:23.992062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_features_mus","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:25:23.994705Z","iopub.execute_input":"2023-05-16T21:25:23.995853Z","iopub.status.idle":"2023-05-16T21:25:24.035132Z","shell.execute_reply.started":"2023-05-16T21:25:23.995795Z","shell.execute_reply":"2023-05-16T21:25:24.033672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_feature = pd.concat([new_features_hum.T, new_features_mus.T])","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:26:00.704243Z","iopub.execute_input":"2023-05-16T21:26:00.70469Z","iopub.status.idle":"2023-05-16T21:26:00.720533Z","shell.execute_reply.started":"2023-05-16T21:26:00.704653Z","shell.execute_reply":"2023-05-16T21:26:00.719356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_feature","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:26:02.922469Z","iopub.execute_input":"2023-05-16T21:26:02.922957Z","iopub.status.idle":"2023-05-16T21:26:02.97431Z","shell.execute_reply.started":"2023-05-16T21:26:02.92291Z","shell.execute_reply":"2023-05-16T21:26:02.973034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ridge regression","metadata":{}},{"cell_type":"markdown","source":"Loading train labels","metadata":{}},{"cell_type":"code","source":"%%time\nn_labels_to_consider = 1499\ntrainTerms = pd.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\",sep=\"\\t\")\nprint(trainTerms.shape)\ndisplay(trainTerms.head(2))\nvec_freqCount = (trainTerms['term'].value_counts())\nprint(vec_freqCount )\n\nprint()\nlabels_to_consider = list(vec_freqCount.index[:n_labels_to_consider] )\nprint('n_labels_to_consider:', len(labels_to_consider), 'First 10:', labels_to_consider[:10] ) \n","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:26:08.170724Z","iopub.execute_input":"2023-05-16T21:26:08.171214Z","iopub.status.idle":"2023-05-16T21:26:11.916731Z","shell.execute_reply.started":"2023-05-16T21:26:08.171168Z","shell.execute_reply":"2023-05-16T21:26:11.915775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load protein IDs","metadata":{"execution":{"iopub.status.busy":"2023-05-16T20:41:34.652359Z","iopub.execute_input":"2023-05-16T20:41:34.652905Z","iopub.status.idle":"2023-05-16T20:41:34.661169Z","shell.execute_reply.started":"2023-05-16T20:41:34.652861Z","shell.execute_reply":"2023-05-16T20:41:34.659566Z"}}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/train_ids.npy'\nvec_train_protein_ids = np.load(fn)\nprint(vec_train_protein_ids.shape)\nvec_train_protein_ids","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:26:14.633693Z","iopub.execute_input":"2023-05-16T21:26:14.63416Z","iopub.status.idle":"2023-05-16T21:26:14.65139Z","shell.execute_reply.started":"2023-05-16T21:26:14.634117Z","shell.execute_reply":"2023-05-16T21:26:14.649845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Prepare Y","metadata":{}},{"cell_type":"code","source":"train_size = 142246\nY = np.zeros( (train_size ,n_labels_to_consider) )\nprint(Y.shape)\n\nseries_train_protein_ids = pd.Series(vec_train_protein_ids ) # \n\ntrainTerms_smaller = trainTerms[ trainTerms['term'].isin( labels_to_consider ) ] # to speed-up the next step \nprint( trainTerms_smaller.shape)\n\nfor i in tqdm(range(Y.shape[1])):\n    m = trainTerms_smaller['term'] ==  labels_to_consider[i]\n#     m.sum()\n    Y[:,i] =  series_train_protein_ids.isin(  set(trainTerms_smaller[m]['EntryID'] ) ).astype(float )\nY \n","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:26:17.18322Z","iopub.execute_input":"2023-05-16T21:26:17.183663Z","iopub.status.idle":"2023-05-16T21:28:42.92477Z","shell.execute_reply.started":"2023-05-16T21:26:17.183622Z","shell.execute_reply":"2023-05-16T21:28:42.923256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading precalculated embedings","metadata":{}},{"cell_type":"code","source":"%%time\n\n# fn = '/kaggle/input/protein-embeddings-1/reduced_embeddings_file.npy'\n# fn = '/kaggle/input/protein-embeddings-1/embed_protbert_train_clip_1200_first_70000_prot.csv'\nfn = '/kaggle/input/t5embeds/train_embeds.npy'\n# fn = '/kaggle/input/t5embeds/test_embeds.npy'\n\nprint(fn)\nif '.csv' in fn:\n    df = pd.read_csv(fn, index_col = 0)\n    X = df.values\nelif '.npy' in fn:\n    X = np.load(fn)\nprint(X.shape)\nX","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:31:45.710102Z","iopub.execute_input":"2023-05-16T21:31:45.710608Z","iopub.status.idle":"2023-05-16T21:31:46.411983Z","shell.execute_reply.started":"2023-05-16T21:31:45.710565Z","shell.execute_reply":"2023-05-16T21:31:46.411069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load protein IDs","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/t5embeds/train_ids.npy'\nvec_train_protein_ids = np.load(fn)\nprint(vec_train_protein_ids.shape)\nvec_train_protein_ids","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:31:52.363644Z","iopub.execute_input":"2023-05-16T21:31:52.364102Z","iopub.status.idle":"2023-05-16T21:31:52.379689Z","shell.execute_reply.started":"2023-05-16T21:31:52.364062Z","shell.execute_reply":"2023-05-16T21:31:52.378514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_ext = []\nfor ind, prot_id in tqdm(enumerate(vec_train_protein_ids)):\n    if prot_id in new_feature.index:\n        X_ext.append(np.r_[X[ind,:], new_feature.loc[prot_id]])\n    else:\n        X_ext.append(np.r_[X[ind,:], [np.mean(X[ind,:])] * 100])","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:28:43.466306Z","iopub.execute_input":"2023-05-16T21:28:43.466668Z","iopub.status.idle":"2023-05-16T21:28:58.014037Z","shell.execute_reply.started":"2023-05-16T21:28:43.466631Z","shell.execute_reply":"2023-05-16T21:28:58.012837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.array(X_ext)","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:32:30.385108Z","iopub.execute_input":"2023-05-16T21:32:30.385538Z","iopub.status.idle":"2023-05-16T21:32:30.983617Z","shell.execute_reply.started":"2023-05-16T21:32:30.385499Z","shell.execute_reply":"2023-05-16T21:32:30.982583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nIX = np.arange(len(X))\nIX_train, IX_test, _,_ = train_test_split( IX, IX, train_size=0.1, random_state=42)\nprint(len(IX_train), len(IX_test),  IX_train[:10], IX_test[:10] )","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:32:34.467051Z","iopub.execute_input":"2023-05-16T21:32:34.468143Z","iopub.status.idle":"2023-05-16T21:32:34.484074Z","shell.execute_reply.started":"2023-05-16T21:32:34.468096Z","shell.execute_reply":"2023-05-16T21:32:34.482864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Modeling","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Ridge\n\nmodel = Ridge(alpha=1.0)\nstr_model_id = 'Ridge1'\n\ndf_models_stat = pd.DataFrame()\nmodel","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:32:44.201058Z","iopub.execute_input":"2023-05-16T21:32:44.202304Z","iopub.status.idle":"2023-05-16T21:32:44.212608Z","shell.execute_reply.started":"2023-05-16T21:32:44.202252Z","shell.execute_reply":"2023-05-16T21:32:44.210976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nimport time\nfrom sklearn.metrics import roc_auc_score\n\nt0 = time.time()\nmodel.fit(X[IX_train,:],Y[IX_train,:])\nY_pred_test = model.predict(X[IX_test,:])\ntt = time.time() - t0\nprint(str_model_id, tt)\nl = []\nfor i in range(Y.shape[1]):\n    if len(np.unique(Y[IX_test,i]) ) > 1:\n        s = roc_auc_score(Y[IX_test,i], Y_pred_test[:,i]);\n    else:\n        s = 0.5\n    l.append(s)        \n    if i %10 == 0:\n        print(i, s)\ndf_models_stat.loc[str_model_id,'RocAuc Mean Test'] = np.mean(l)\ndf_models_stat.loc[str_model_id,'Time'] = np.round(tt,1)\ndf_models_stat.loc[str_model_id,'Test Size'] = len(IX_test)\ndf_models_stat","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:32:46.841702Z","iopub.execute_input":"2023-05-16T21:32:46.842181Z","iopub.status.idle":"2023-05-16T21:33:05.126166Z","shell.execute_reply.started":"2023-05-16T21:32:46.842136Z","shell.execute_reply":"2023-05-16T21:33:05.124659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Scores statistics across targets","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.hist(l)\nplt.show()\npd.Series(l).describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-16T21:33:05.128655Z","iopub.execute_input":"2023-05-16T21:33:05.129063Z","iopub.status.idle":"2023-05-16T21:33:05.316077Z","shell.execute_reply.started":"2023-05-16T21:33:05.129023Z","shell.execute_reply":"2023-05-16T21:33:05.314872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Full dataset model","metadata":{}},{"cell_type":"code","source":"%%time\nmodel.fit(X,Y)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"jupyter":{"source_hidden":true}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}