{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ? \n\nExplore modeling in different embeddings datasets \n\n\n## Versions:\n\n#### 1 esm2large - best alpha = 1000 and then it it is even better than t5\n","metadata":{}},{"cell_type":"markdown","source":"# Key params","metadata":{}},{"cell_type":"code","source":"n_samples_to_consider = 1_000_000\nn_labels_to_consider = 1850\n","metadata":{"execution":{"iopub.status.busy":"2023-07-16T11:57:12.85826Z","iopub.execute_input":"2023-07-16T11:57:12.859178Z","iopub.status.idle":"2023-07-16T11:57:12.864115Z","shell.execute_reply.started":"2023-07-16T11:57:12.859139Z","shell.execute_reply":"2023-07-16T11:57:12.863005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparataion and data load ","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport time\nt0start = time.time()\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\ntry: # for TPU one should install it\n    import matplotlib.pyplot as plt\n    import seaborn as sns # there is no one for TPU machine \nexcept:\n    print('matplotlib or seaborn  is not installed')\n    pass\n    \n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-16T11:57:12.867908Z","iopub.execute_input":"2023-07-16T11:57:12.868963Z","iopub.status.idle":"2023-07-16T11:57:12.895656Z","shell.execute_reply.started":"2023-07-16T11:57:12.8689Z","shell.execute_reply":"2023-07-16T11:57:12.894558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_emb_fn = os.listdir('/kaggle/input/protein-embeddings-4-cafa5-selected/Embeddings_43k/')\nprint(len(list_emb_fn))\nprint(list_emb_fn)","metadata":{"execution":{"iopub.status.busy":"2023-07-16T11:57:12.898098Z","iopub.execute_input":"2023-07-16T11:57:12.89878Z","iopub.status.idle":"2023-07-16T11:57:12.90525Z","shell.execute_reply.started":"2023-07-16T11:57:12.89874Z","shell.execute_reply":"2023-07-16T11:57:12.904147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n######################################################################################3\n###############  Load Target Values\n######################################################################################3\n\n# # import scipy\n# import scipy.sparse\n# fn = '/kaggle/input/protein-embeddings-4-cafa5-selected/Y31466_sparse_float32_cut43k.npz'\n# print(fn)\n# Y = scipy.sparse.load_npz(fn)\n# print(Y.shape)\n# Y = Y[:,:n_labels_to_consider].toarray()\n# print(Y.shape)\n# print(Y[:2,:3] )\n\nfn = '/kaggle/input/protein-embeddings-4-cafa5-selected/Y1850_cut43k.npy'\nY = np.load(fn)\nprint(Y.shape)\nY = Y[:n_samples_to_consider,:n_labels_to_consider]\nprint(Y.shape)\nprint(Y[:2,:3] )\n\n######################################################################################3\n###############  Load Y labels\n######################################################################################3\nprint()\nfn = '/kaggle/input/protein-embeddings-4-cafa5-selected/Y1850_terms_ids.csv'\nY_labels = pd.read_csv(fn).iloc[:n_labels_to_consider,:]\nprint('Y_labels', Y_labels.shape)\ndisplay(Y_labels.head(3) )\nprint()\nlabels_to_consider = Y_labels.iloc[:,0].values\nprint('labels_to_consider', len(labels_to_consider), labels_to_consider[:15])  ","metadata":{"execution":{"iopub.status.busy":"2023-07-16T11:57:12.906362Z","iopub.execute_input":"2023-07-16T11:57:12.906684Z","iopub.status.idle":"2023-07-16T11:57:13.043073Z","shell.execute_reply.started":"2023-07-16T11:57:12.906656Z","shell.execute_reply":"2023-07-16T11:57:13.041896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n######################################################################################3\n###############  Load Train Ids\n######################################################################################3\n\nprint()\nfn = '/kaggle/input/protein-embeddings-4-cafa5-selected/train_ids_cut43k.npy'\nvec_train_protein_ids = np.load(fn)\nprint('vec_train_protein_ids:', vec_train_protein_ids.shape)\nprint( vec_train_protein_ids[:10] )\nvec_train_protein_ids = vec_train_protein_ids[:n_samples_to_consider]\n","metadata":{"execution":{"iopub.status.busy":"2023-07-16T11:57:13.045865Z","iopub.execute_input":"2023-07-16T11:57:13.046428Z","iopub.status.idle":"2023-07-16T11:57:13.064915Z","shell.execute_reply.started":"2023-07-16T11:57:13.046374Z","shell.execute_reply":"2023-07-16T11:57:13.063915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n######################################################################################3\n###############  Load Folds\n######################################################################################3\n\n\n# fn = '/kaggle/input/cafa5-data-selected/embeds_43k_Y1850_etc/df_folds_43k_Nsims6_evalue0d001.csv'\nfn = '/kaggle/input/protein-embeddings-4-cafa5-selected/df_folds_strict_Nsims6_evalue0d001_cut43k.csv'\ndf_folds = pd.read_csv(fn,index_col = 0)\ndf_folds = df_folds.iloc[:n_samples_to_consider,:]\nn_samples_to_consider = len(df_folds) # In case n_samples_to_consider is greater than actual sample number we cut it down \nprint('df_folds', df_folds.shape )\ndisplay(df_folds.head(3) )\n\ntry:\n    print('Check indexes coincide, expect zero: ',  (vec_train_protein_ids != df_folds.index.values).sum() )\nexcept:\n    pass","metadata":{"execution":{"iopub.status.busy":"2023-07-16T11:57:13.066575Z","iopub.execute_input":"2023-07-16T11:57:13.067032Z","iopub.status.idle":"2023-07-16T11:57:13.17591Z","shell.execute_reply.started":"2023-07-16T11:57:13.066995Z","shell.execute_reply":"2023-07-16T11:57:13.174877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n######################################################################################3\n###############  Load trainTerms\n######################################################################################3\n\nprint()\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv'\nprint(fn)\ntrainTerms = pd.read_csv(fn, sep=\"\\t\")\nprint(trainTerms.shape)\ndisplay(trainTerms.head(3))","metadata":{"execution":{"iopub.status.busy":"2023-07-16T11:57:13.177771Z","iopub.execute_input":"2023-07-16T11:57:13.178191Z","iopub.status.idle":"2023-07-16T11:57:16.059656Z","shell.execute_reply.started":"2023-07-16T11:57:13.178152Z","shell.execute_reply":"2023-07-16T11:57:16.058242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metric code ","metadata":{}},{"cell_type":"code","source":"# Evaluation for CAFA \n# https://github.com/BioComputingUP/CAFA-evaluator\n\nflag_correct_metric_computation_bug_found_by_Anton = True\n# Use corrected version of code - as proposed by Anton Vakhrushev ( Btbpanda )\n# See: https://www.kaggle.com/competitions/cafa-5-protein-function-prediction/discussion/420241\n\n# Code is also adapted to work on Kaggle, and speeded-up by use of sparse matrices \n\nimport numpy as np\nimport pandas as pd\nimport multiprocessing as mp\nimport copy\nimport logging\n\nclass Graph:\n    \"\"\"\n    Ontology class. One ontology == one namespace\n    DAG is the adjacence matrix (sparse) which represent a Directed Acyclic Graph where\n    DAG(i,j) == 1 means that the go term i is_a (or is part_of) j\n    Parents that are in a different namespace are discarded\n    \"\"\"\n    def __init__(self, namespace, terms_dict, ia_dict=None, orphans=False):\n        \"\"\"\n        terms_dict = {term: {name: , namespace: , def: , alt_id: , rel:}}\n        \"\"\"\n        self.namespace = namespace\n        self.dag = []  # [[], ...] terms (rows, axis 0) x parents (columns, axis 1)\n        self.terms_dict = {}  # {term: {index: , name: , namespace: , def: }  used to assign term indexes in the gt\n        self.terms_list = []  # [{id: term, name:, namespace: , def:, adg: [], children: []}, ...]\n        self.idxs = None  # Number of terms\n        self.order = None\n        self.toi = None\n        self.ia = None\n\n        rel_list = []\n        for self.idxs, (term_id, term) in enumerate(terms_dict.items()):\n            rel_list.extend([[term_id, rel, term['namespace']] for rel in term['rel']])\n            self.terms_list.append({'id': term_id, 'name': term['name'], 'namespace': namespace, 'def': term['def'],\n                                 'adj': [], 'children': []})\n            self.terms_dict[term_id] = {'index': self.idxs, 'name': term['name'], 'namespace': namespace, 'def': term['def']}\n            for a_id in term['alt_id']:\n                self.terms_dict[a_id] = copy.copy(self.terms_dict[term_id])\n        self.idxs += 1\n\n        self.dag = np.zeros((self.idxs, self.idxs), dtype='bool')\n\n        # id1 term (row, axis 0), id2 parent (column, axis 1)\n        for id1, id2, ns in rel_list:\n            if self.terms_dict.get(id2):\n                i = self.terms_dict[id1]['index']\n                j = self.terms_dict[id2]['index']\n                self.dag[i, j] = 1\n                self.terms_list[i]['adj'].append(j)\n                self.terms_list[j]['children'].append(i)\n                logging.debug(\"i,j {},{} {},{}\".format(i, j, id1, id2))\n            else:\n                logging.debug('Skipping branch to external namespace: {}'.format(id2))\n        logging.debug(\"dag {}\".format(self.dag))\n        # Topological sorting\n        self.top_sort()\n        logging.debug(\"order sorted {}\".format(self.order))\n\n        if orphans:\n            self.toi = np.arange(self.dag.shape[0])  # All terms, also those without parents\n        else:\n            self.toi = np.nonzero(self.dag.sum(axis=1) > 0)[0]  # Only terms with parents\n        logging.debug(\"toi {}\".format(self.toi))\n\n        if ia_dict is not None:\n            self.set_ia(ia_dict)\n\n        return\n\n    def top_sort(self):\n        \"\"\"\n        Takes a sparse matrix representing a DAG and returns an array with nodes indexes in topological order\n        https://en.wikipedia.org/wiki/Topological_sorting\n        \"\"\"\n        indexes = []\n        visited = 0\n        (rows, cols) = self.dag.shape\n\n        # create a vector containing the in-degree of each node\n        in_degree = self.dag.sum(axis=0)\n        # logging.debug(\"degree {}\".format(in_degree))\n\n        # find the nodes with in-degree 0 (leaves) and add them to the queue\n        queue = np.nonzero(in_degree == 0)[0].tolist()\n        # logging.debug(\"queue {}\".format(queue))\n\n        # for each element of the queue increment visits, add them to the list of ordered nodes\n        # and decrease the in-degree of the neighbor nodes\n        # and add them to the queue if they reach in-degree == 0\n        while queue:\n            visited += 1\n            idx = queue.pop(0)\n            indexes.append(idx)\n            in_degree[idx] -= 1\n            l = self.terms_list[idx]['adj']\n            if len(l) > 0:\n                for j in l:\n                    in_degree[j] -= 1\n                    if in_degree[j] == 0:\n                        queue.append(j)\n\n        # if visited is equal to the number of nodes in the graph then the sorting is complete\n        # otherwise the graph can't be sorted with topological order\n        if visited == rows:\n            self.order = indexes\n        else:\n            raise Exception(\"The sparse matrix doesn't represent an acyclic graph\")\n\n    def set_ia(self, ia_dict):\n        self.ia = np.zeros(self.idxs, dtype='float')\n        for term_id in self.terms_dict:\n            if ia_dict.get(term_id):\n                self.ia[self.terms_dict[term_id]['index']] = ia_dict.get(term_id)\n            else:\n                logging.debug('Missing IA for term: {}'.format(term_id))\n        # Convert inf to zero\n        np.nan_to_num(self.ia, copy=False, nan=0, posinf=0, neginf=0)\n        self.toi = np.nonzero(self.ia > 0)[0]\n\n\nclass Prediction:\n    \"\"\"\n    The score matrix contains the scores given by the predictor for every node of the ontology\n    \"\"\"\n    def __init__(self, ids, matrix, idx, namespace=None):\n        self.ids = ids\n        self.matrix = matrix  # scores\n        self.next_idx = idx\n        # self.n_pred_seq = idx + 1\n        self.namespace = namespace\n\n    def __str__(self):\n        return \"\\n\".join([\"{}\\t{}\\t{}\".format(index, self.matrix[index], self.namespace) for index, _id in enumerate(self.ids)])\n\n\nclass GroundTruth:\n    def __init__(self, ids, matrix, namespace=None):\n        self.ids = ids\n        self.matrix = matrix\n        self.namespace = namespace\n\n\ndef propagate(matrix, ont, order, mode='max'):\n    \"\"\"\n    Update inplace the score matrix (proteins x terms) up to the root taking the max between children and parents\n    \"\"\"\n    if matrix.shape[0] == 0:\n        raise Exception(\"Empty matrix\")\n\n    deepest = np.where(np.sum(matrix[:, order], axis=0) > 0)[0][0]\n    if deepest.size == 0:\n        raise Exception(\"The matrix is empty\")\n\n    # Remove leaves\n    order_ = np.delete(order, [range(0, deepest)])\n\n    for i in order_:\n        # Get direct children\n        children = np.where(ont.dag[:, i] != 0)[0]\n        if children.size > 0:\n            cols = np.concatenate((children, [i]))\n            if mode == 'max':\n                matrix[:, i] = matrix[:, cols].max(axis=1)\n            elif mode == 'fill':\n                rows = np.where(matrix[:, i] == 0)[0]\n                if rows.size:\n                    idx = np.ix_(rows, cols)\n                    if flag_correct_metric_computation_bug_found_by_Anton:\n                        # Corrected way (see https://www.kaggle.com/competitions/cafa-5-protein-function-prediction/discussion/420241 )\n                        matrix[rows, i] = matrix[idx].max(axis=1) #  matrix[idx].max(axis=1)[0] # Correction: https://www.kaggle.com/competitions/cafa-5-protein-function-prediction/discussion/420241\n                    else:\n                        # Old way - not corrected\n                        matrix[rows, i] = matrix[idx].max(axis=1)[0] # Correction: https://www.kaggle.com/competitions/cafa-5-protein-function-prediction/discussion/420241\n                        \n    return\n\n\ndef obo_parser(obo_file, valid_rel=(\"is_a\", \"part_of\")):\n    \"\"\"\n    Parse a OBO file and returns a list of ontologies, one for each namespace.\n    Obsolete terms are excluded as well as external namespaces.\n    \"\"\"\n    term_dict = {}\n    term_id = None\n    namespace = None\n    name = None\n    term_def = None\n    alt_id = []\n    rel = []\n    obsolete = True\n    with open(obo_file) as f:\n        for line in f:\n            line = line.strip().split(\": \")\n            if line and len(line) > 1:\n                k = line[0]\n                v = \": \".join(line[1:])\n                if k == \"id\":\n                    # Populate the dictionary with the previous entry\n                    if term_id is not None and obsolete is False and namespace is not None:\n                        term_dict.setdefault(namespace, {})[term_id] = {'name': name,\n                                                                       'namespace': namespace,\n                                                                       'def': term_def,\n                                                                       'alt_id': alt_id,\n                                                                       'rel': rel}\n                    # Assign current term ID\n                    term_id = v\n\n                    # Reset optional fields\n                    alt_id = []\n                    rel = []\n                    obsolete = False\n                    namespace = None\n\n                elif k == \"alt_id\":\n                    alt_id.append(v)\n                elif k == \"name\":\n                    name = v\n                elif k == \"namespace\" and v != 'external':\n                    namespace = v\n                elif k == \"def\":\n                    term_def = v\n                elif k == 'is_obsolete':\n                    obsolete = True\n                elif k == \"is_a\" and k in valid_rel:\n                    s = v.split('!')[0].strip()\n                    rel.append(s)\n                elif k == \"relationship\" and v.startswith(\"part_of\") and \"part_of\" in valid_rel:\n                    s = v.split()[1].strip()\n                    rel.append(s)\n\n        # Last record\n        if obsolete is False and namespace is not None:\n            term_dict.setdefault(namespace, {})[term_id] = {'name': name,\n                                                          'namespace': namespace,\n                                                          'def': term_def,\n                                                          'alt_id': alt_id,\n                                                          'rel': rel}\n    return term_dict\n\n\ndef gt_parser(gt_file, ontologies):\n    \"\"\"\n    Parse ground truth file. Discard terms not included in the ontology.\n    \"\"\"\n    gt_dict = {}\n    with open(gt_file) as f:\n        for line in f:\n            line = line.strip().split()\n            if line:\n                p_id, term_id = line[:2]\n                for ont in ontologies:\n                    if term_id in ont.terms_dict:\n                        gt_dict.setdefault(ont.namespace, {}).setdefault(p_id, []).append(term_id)\n                        break\n\n    gts = {}\n    for ont in ontologies:\n        if gt_dict.get(ont.namespace):\n            matrix = np.zeros((len(gt_dict[ont.namespace]), ont.idxs), dtype='bool')\n            ids = {}\n            for i, p_id in enumerate(gt_dict[ont.namespace]):\n                ids[p_id] = i\n                for term_id in gt_dict[ont.namespace][p_id]:\n                    matrix[i, ont.terms_dict[term_id]['index']] = 1\n            propagate(matrix, ont, ont.order, mode='max')\n            gts[ont.namespace] = GroundTruth(ids, matrix, ont.namespace)\n\n    return gts\n\n\ndef pred_parser(f, ontologies, gts, prop_mode, max_terms=None):\n    \"\"\"\n    Parse a prediction file and returns a list of prediction objects, one for each namespace.\n    If a predicted is predicted multiple times for the same target, it stores the max.\n    This is the slow step if the input file is huge, ca. 1 minute for 5GB input on SSD disk.\n    \"\"\"\n    ids = {}\n    matrix = {}\n    ns_dict = {}  # {namespace: term}\n    onts = {ont.namespace: ont for ont in ontologies}\n    for ns in gts:\n        matrix[ns] = np.zeros(gts[ns].matrix.shape, dtype='float')\n        ids[ns] = {}\n        for term in onts[ns].terms_dict:\n            ns_dict[term] = ns\n\n    for line in f:\n        p_id, term_id, prob = line\n        ns = ns_dict.get(term_id)\n        if ns in gts and p_id in gts[ns].ids:\n            i = gts[ns].ids[p_id]\n            if max_terms is None or np.count_nonzero(matrix[ns][i]) <= max_terms:\n                j = onts[ns].terms_dict.get(term_id)['index']\n                ids[ns][p_id] = i\n                matrix[ns][i, j] = max(matrix[ns][i, j], float(prob))\n\n    predictions = []\n    for ns in ids:\n        if ids[ns]:\n            propagate(matrix[ns], onts[ns], onts[ns].order, mode=prop_mode)\n            predictions.append(Prediction(ids[ns], matrix[ns], len(ids[ns]), ns))\n\n    if not predictions:\n        raise Exception(\"Empty prediction, check format\")\n\n    return predictions\n\n\ndef ia_parser(file):\n    ia_dict = {}\n    with open(file) as f:\n        for line in f:\n            if line:\n                term, ia = line.strip().split()\n                ia_dict[term] = float(ia)\n    return ia_dict\n\n# Computes the root terms in the dag\ndef get_roots_idx(dag):\n    return np.where(dag.sum(axis=1) == 0)[0]\n\n\n# Computes the leaf terms in the dag\ndef get_leafs_idx(dag):\n    return np.where(dag.sum(axis=0) == 0)[0]\n\n\n# Return a mask for all the predictions (matrix) >= tau\ndef solidify_prediction(pred, tau):\n    return pred >= tau\n\n\n# computes the f metric for each precision and recall in the input arrays\ndef compute_f(pr, rc):\n    n = 2 * pr * rc\n    d = pr + rc\n    return np.divide(n, d, out=np.zeros_like(n, dtype=float), where=d != 0)\n\n\ndef compute_s(ru, mi):\n    return np.sqrt(ru**2 + mi**2)\n    # return np.where(np.isnan(ru), mi, np.sqrt(ru + np.nan_to_num(mi)))\n\nimport time\nfrom scipy.sparse import csr_matrix\n\ndef compute_metrics_(tau_arr, g, pred, toi, n_gt, wn_gt=None, ic_arr=None):\n\n    verbose = 0;         \n\n    if verbose >= 10:\n        t0 = time.time()\n    \n    metrics = np.zeros((len(tau_arr), 7), dtype='float')  # cov, pr, rc, wpr, wrc, ru, mi\n\n    if verbose >= 10:\n        print('type(toi), toi', type(toi), toi )\n    tmp = pred.matrix[:, toi]\n    if verbose >= 10:\n        print('type(tmp), tmp.shape', type(tmp), tmp.shape )\n    p_s = csr_matrix(tmp )\n    ic_arr_toi = ic_arr[toi]\n    if verbose >= 10:\n        print('type(ic_arr_toi), ic_arr_toi.shape', type(ic_arr_toi), ic_arr_toi.shape )\n\n    \n    g_s = csr_matrix( g )\n    \n    if verbose >= 10:\n        print( 'csr_matrix done %.1f'%(time.time( ) - t0 ), 'p_s.shape, g_s.shape:', p_s.shape, g_s.shape )    \n\n\n    \n    for i, tau in enumerate(tau_arr):\n\n#         p = solidify_prediction(pred_matrix_toi, tau)\n\n#         # number of proteins with at least one term predicted with score >= tau\n#         metrics[i, 0] = (p.sum(axis=1) > 0).sum()\n\n#         # Terms subsets\n#         intersection = np.logical_and(p, g)  # TP                                # SLOW PART !!!\n\n#         # Subsets size\n#         n_pred = p.sum(axis=1)\n#         n_intersection = intersection.sum(axis=1)\n\n#         # Precision, recall\n#         metrics[i, 1] = np.divide(n_intersection, n_pred, out=np.zeros_like(n_intersection, dtype='float'),\n#                                   where=n_pred > 0).sum()\n#         metrics[i, 2] = np.divide(n_intersection, n_gt, out=np.zeros_like(n_gt, dtype='float'), where=n_gt > 0).sum()\n\n\n        if verbose >= 100:\n            t0 = time.time()\n            print()\n            print(i, tau, 'Start %.1f'%(time.time( ) - t0 ) )\n        p = p_s > tau # solidify_prediction(p, tau)\n        if verbose >= 100:\n            print(i, tau, 'solidify done %.1f'%(time.time( ) - t0 ), 'p.shape:', p.shape,  )\n#         p_s = csr_matrix(p)\n#         print(i, tau, 'csr_matrix done %.1f'%(time.time( ) - t0 ), 'p.shape:', p.shape,  )\n        \n#         print(p.shape)\n        # number of proteins with at least one term predicted with score >= tau\n        metrics[i, 0] = (p.sum(axis=1) > 0).sum()\n\n        # Terms subsets\n#         intersection = np.logical_and(p, g)  # TP\n        intersection = p.multiply( g_s)  # TP\n\n\n        if ic_arr is not None:\n            \n            # Weighted precision, recall\n            # wn_pred = (p * ic_arr_toi).sum(axis=1) # \n#             wn_pred = np.dot(p , ic_arr_toi)\n            wn_pred = p.dot( ic_arr_toi)\n            # wn_intersection = (intersection *ic_arr_toi).sum(axis=1)\n#             wn_intersection = np.dot( intersection , ic_arr_toi )\n            wn_intersection =  intersection.dot( ic_arr_toi )\n            \n            if verbose >= 100:\n                print(i, tau, 'After w_pred wn_intersection  %.1f'%(time.time( ) - t0 ) )\n            \n            metrics[i, 3] = np.divide(wn_intersection, wn_pred, out=np.zeros( wn_intersection.shape, dtype='float'),\n                                      where=wn_pred > 0).sum()\n            metrics[i, 4] = np.divide(wn_intersection, wn_gt, out=np.zeros(wn_intersection.shape, dtype='float'),\n                                      where=n_gt > 0).sum()\n            if verbose >= 100:\n                print(i, tau, 'After metrics 3,4   %.1f'%(time.time( ) - t0 ) )\n\n#             # Terms subsets\n#             remaining = np.logical_and(np.logical_not(p), g)  # FN --> not predicted but in the ground truth\n#             mis = np.logical_and(p, np.logical_not(g))  # FP --> predicted but not in the ground truth\n\n#             print(i, tau, 'After remining and miss  %.1f'%(time.time( ) - t0 ) )\n            \n#             # Misinformation, remaining uncertainty\n#             metrics[i, 5] = (remaining * ic_arr_toi).sum(axis=1).sum()\n#             metrics[i, 6] = (mis * ic_arr_toi).sum(axis=1).sum()\n\n    return metrics\n\n\ndef compute_metrics(pred, gt, toi, tau_arr, ic_arr=None, n_cpu=0):\n    \"\"\"\n    Takes the prediction and the ground truth and for each threshold in tau_arr\n    calculates the confusion matrix and returns the coverage,\n    precision, recall, remaining uncertainty and misinformation.\n    Toi is the list of terms (indexes) to be considered\n    \"\"\"\n    g = gt.matrix[:, toi]\n    n_gt = g.sum(axis=1)\n    wn_gt = None\n    if ic_arr is not None:\n        wn_gt = (g * ic_arr[toi]).sum(axis=1)\n\n    # Parallelization\n    if n_cpu == 0:\n        n_cpu = mp.cpu_count()\n\n    arg_lists = [[tau_arr, g, pred, toi, n_gt, wn_gt, ic_arr] for tau_arr in np.array_split(tau_arr, n_cpu)]\n    if 0:\n        # Original parallel way (# It does not work on Kaggle)\n        arg_lists = [[tau_arr, g, pred, toi, n_gt, wn_gt, ic_arr] for tau_arr in np.array_split(tau_arr, n_cpu)]\n        with mp.Pool(processes=n_cpu) as pool:\n            metrics = np.concatenate(pool.starmap(compute_metrics_, arg_lists), axis=0)\n    else: \n        # no-parallel: \n        metrics = compute_metrics_(tau_arr, g, pred, toi, n_gt, wn_gt, ic_arr )\n\n    return pd.DataFrame(metrics, columns=[\"cov\", \"pr\", \"rc\", \"wpr\", \"wrc\", \"ru\", \"mi\"])\n\n\ndef evaluate_prediction(prediction, gt, ontologies, tau_arr, normalization='cafa', n_cpu=0):\n    dfs = []\n    for p in prediction:\n        ns = p.namespace\n        ne = np.full(len(tau_arr), gt[ns].matrix.shape[0])\n\n        ont = [o for o in ontologies if o.namespace == ns][0]\n\n        # cov, pr, rc, wpr, wrc, ru, mi\n        metrics = compute_metrics(p, gt[ns], ont.toi, tau_arr, ont.ia, n_cpu)\n\n        for column in [\"pr\", \"rc\", \"wpr\", \"wrc\", \"ru\", \"mi\"]:\n            if normalization == 'gt' or (column in [\"rc\", \"wrc\"] and normalization == 'cafa'):\n                metrics[column] = np.divide(metrics[column], ne, out=np.zeros_like(metrics[column], dtype='float'), where=ne > 0)\n            else:\n                metrics[column] = np.divide(metrics[column], metrics[\"cov\"], out=np.zeros_like(metrics[column], dtype='float'), where=metrics[\"cov\"] > 0)\n\n        metrics['ns'] = [ns] * len(tau_arr)\n        metrics['tau'] = tau_arr\n        metrics['cov'] = np.divide(metrics['cov'], ne, out=np.zeros_like(metrics['cov'], dtype='float'), where=ne > 0)\n        metrics['f'] = compute_f(metrics['pr'], metrics['rc'])\n        metrics['wf'] = compute_f(metrics['wpr'], metrics['wrc'])\n        metrics['s'] = compute_s(metrics['ru'], metrics['mi'])\n\n        dfs.append(metrics)\n\n    return pd.concat(dfs)\n\n\n# Tau array, used to compute metrics at different score thresholds\nth_step = 0.01\ntau_arr = np.arange(0.01, 1, th_step)\n#Consider terms without parents, e.g. the root(s), in the evaluation\nno_orphans = False\n# Parse and set information accretion (optional)\nia_dict = ia_parser('/kaggle/input/cafa-5-protein-function-prediction/IA.txt')\n\n# Parse the OBO file and creates a different graph for each namespace\nontologies = []\nobo_file = '/kaggle/input/cafa-5-protein-function-prediction/Train/go-basic.obo'\nfor ns, terms_dict in obo_parser(obo_file).items():\n    ontologies.append(Graph(ns, terms_dict, ia_dict, not no_orphans))","metadata":{"execution":{"iopub.status.busy":"2023-07-16T11:57:16.061576Z","iopub.execute_input":"2023-07-16T11:57:16.062326Z","iopub.status.idle":"2023-07-16T11:57:19.568101Z","shell.execute_reply.started":"2023-07-16T11:57:16.062292Z","shell.execute_reply":"2023-07-16T11:57:19.56691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metric wrappers and other statistics/scoring functions ","metadata":{}},{"cell_type":"markdown","source":"## F1-metric wrapper \n\nThe code prepares necessary files and data to call core metric evalution. ","metadata":{}},{"cell_type":"code","source":"%%time\nimport os.path\n\ndef get_F1_etc_scores_official_CAFA_evaluation( Y_pred, IX, cutoff_threshold_low = 0.01,    make_plots = True ,  verbose = 0 ): \n    # Input: \n    '''\n    Computation of F1-weighted scores are called here. \n    Here we prepare Y_pred, Y in format required by functions provided by organizers - see github: https://github.com/BioComputingUP/CAFA-evaluator\n    Y_pred  -  predictions\n    trainTerms # initial file with training labels \n    IX  - indices selecting part which correspond to Y_pred in full Y \n    Params:\n    cutoff_threshold_low - predictions lower (strictly) will be dropped (effectively set to zero)\n    make_plots = True \n    verbose = 0\n    \n    Function uses external variables: \n    trainTerms - training labels provided by orgs: /kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\n    vec_train_protein_ids - ids of the proteins in the current train - should correspond to \"X\" - features \n    ontologies - data from: /kaggle/input/cafa-5-protein-function-prediction/Train/go-basic.obo\n    tau_arr - array of thresholds \n    '''\n\n    # Params:\n\n    t00 = time.time()\n    if verbose >= 100:\n        print('Scoring starts. n_samples:', len(IX) )\n\n    ##########################################################################################\n    # Prepare \"ground truth\" - \"gt\" terms(labels) in required format  \n    ##########################################################################################\n\n    # First save to file, because function \"gt_parser\" works with files as input \n    # Only part corresponding to providex indices IX will be generated \n    t0 = time.time()\n    trainTerms[ trainTerms.EntryID.isin(vec_train_protein_ids[IX]) ].to_csv('valid.tsv', sep='\\t', index=False) # Wall time: 4.11 s  for 28k samples\n    if verbose >= 1000:\n        print('save valid.csv %.1f'%(time.time() - t0 )) \n\n    # Prepare \"gt\" labels \n    t0 = time.time()\n    gt = gt_parser('valid.tsv', ontologies) # Wall time: 1min 22s  for 28k samples\n    if verbose >= 100:\n        print('gt_parser %.1f'%(time.time() - t0 ))\n\n    ##########################################################################################\n    # prepare predicitons as list of triples - (protein, term(label), prediction) \n    ##########################################################################################\n\n    t0 = time.time()\n    vec_train_protein_ids_loc = vec_train_protein_ids[IX]\n    preds = []\n    for i in range(len(vec_train_protein_ids_loc)):\n        for j in range(len(labels_to_consider)):\n            if Y_pred[i,j] >= cutoff_threshold_low:\n                preds.append((vec_train_protein_ids_loc[i], \n                              labels_to_consider[j],\n                              Y_pred[i,j]                        ))\n    if verbose >= 1000:            \n        print('create preds %.1f'%(time.time() - t0 ))       \n\n    ##########################################################################################\n    # Parse predictions - propagation happens here  \n    ##########################################################################################\n    t0 = time.time()\n    preds = pred_parser(preds, ontologies, gt, prop_mode='fill', max_terms=500) # \n    if verbose >= 1000:            \n        print('pred_parser %.1f'%(time.time() - t0 ), 'len(preds)', len(preds) )            \n\n    gc.collect()\n\n    ##########################################################################################\n    # Main scores calculations happends here: \n    ##########################################################################################\n    # %%time\n    t0 = time.time()\n    df_metrics = evaluate_prediction(preds, gt, ontologies, tau_arr, n_cpu=1) # Wall time: 37.7 s for 28k samples\n    if verbose >= 1000:            \n        print('evaluate_prediction %.1f'%(time.time() - t0 ), 'got df_metrics with shape:', df_metrics.shape )            \n    if verbose >= 10000:            \n        display( df_metrics.head(2) )\n\n    ##########################################################################################\n    # Comptutations finished. Below are optional plots, output preparartions etc.  \n    ##########################################################################################\n    if verbose >= 100:\n        _t = df_metrics.groupby('ns').agg({'wf':'max'})\n        display( _t )\n        print( _t.mean() ) \n\n    if verbose >= 100:\n        print('F1-scoring finished. %.1f secs passed'%(time.time() - t00 ))\n\n    # %%time\n    if make_plots:\n        list_uv = list(df_metrics['ns'].unique() )\n        #print(list_uv)\n        fig = plt.figure(figsize = (20,4))\n        i0 = 0;\n        for  col in  ['wf', 'wpr', 'wrc' ] :\n            i0+=1\n#             print(i0,col)\n            fig.add_subplot(1,3,i0)\n\n            for uv in list_uv:\n                mask = df_metrics['ns'] == uv\n                v = df_metrics[mask][col]\n                plt.plot(v.values, label = uv)\n            plt.title(col, fontsize  = 20)\n            plt.legend()\n            plt.grid()\n        plt.show()        \n    \n    ########################################################################################\n    # Prepare output of scores : \n    ########################################################################################\n    _t = {'cellular_component':'CCO', 'biological_process':'BPO','molecular_function':'MFO'}\n    dict_scores_etc = {}\n    df_s = df_metrics.groupby('ns').agg({'wf':'max'})\n    dict_scores_etc['F1w'] = np.round( df_s.mean().iloc[0], 6) \n    for k in _t:\n        k2 = _t[k]\n        # print(k,dict_scores_etc )\n        if k in  df_s.index:\n            dict_scores_etc['F1 '+ k2 ] = np.round( df_s.loc[k].iat[0], 6) \n        else:\n            dict_scores_etc['F1 '+ k2 ] = 0\n\n    ########################################################################################\n    # Prepare output of thresholds : \n    ########################################################################################\n    for k in _t:\n        k2 = _t[k]\n        m = df_metrics['ns'] == k\n        if m.sum()>0:\n            IX = df_metrics[m]['wf'].argmax()\n            thres_optimal = df_metrics[m]['tau'].iat[IX]\n            dict_scores_etc['thres '+ k2 ] = thres_optimal\n        else:\n            dict_scores_etc['thres '+ k2 ] = 0\n            \n    dict_scores_etc['F-Scores Time'] = np.round( time.time() - t00   ,1)         \n    if verbose >= 100:\n        print('Scores: ', dict_scores_etc)        \n\n    if os.path.isfile('valid.tsv') :\n        os.remove('valid.tsv')\n        \n    return  dict_scores_etc      ","metadata":{"execution":{"iopub.status.busy":"2023-07-16T11:57:19.571469Z","iopub.execute_input":"2023-07-16T11:57:19.571831Z","iopub.status.idle":"2023-07-16T11:57:19.598387Z","shell.execute_reply.started":"2023-07-16T11:57:19.571798Z","shell.execute_reply":"2023-07-16T11:57:19.597336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Main scoring function called from the main code\n\nIt saves various statistics. \n\nIt calls F1-metric wrapper","metadata":{}},{"cell_type":"code","source":"%%time\ndef update_scores(df_stat, str_full_id, fold_id,  Y ,  Y_pred,IX, cutoff_threshold_low = 0.01, fit_time = None,   predict_time = None , blend_mode = None ,\n                          str_cv_info = None ):\n    IX_df_stat = len(df_stat) + 1 \n    t0 = time.time()\n    df_stat.loc[IX_df_stat,'FullId'] = str_full_id\n    df_stat.loc[IX_df_stat,'RocAuc ravel'] = np.round( roc_auc_score (Y.ravel(),Y_pred.ravel()  ) , 4)\n\n    dict_scores_etc = get_F1_etc_scores_official_CAFA_evaluation( Y_pred, IX,  cutoff_threshold_low = cutoff_threshold_low ,     make_plots = True ,  verbose = 10000 )\n    for k in dict_scores_etc:\n        df_stat.loc[IX_df_stat,k] = dict_scores_etc[k]\n        \n    df_stat.loc[IX_df_stat,'Fold'] = fold_id\n\n    df_stat.loc[IX_df_stat,'Scoring time'] = np.round( time.time() - t0,1 )\n    df_stat.loc[IX_df_stat,'CV'] = str_cv_info\n    df_stat.loc[IX_df_stat,'Fit time'] = np.round( fit_time , 1)\n    df_stat.loc[IX_df_stat,'Predict time'] = np.round( predict_time , 1)\n    df_stat.loc[IX_df_stat,'n_targets'] = Y.shape[1]\n    df_stat.loc[IX_df_stat,'n_samples'] = Y.shape[0]\n    df_stat.loc[IX_df_stat,'blend_mode'] = blend_mode\n    \n    df_foldwise_stat.to_csv('df_foldwise_stat.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-16T11:57:19.599979Z","iopub.execute_input":"2023-07-16T11:57:19.600377Z","iopub.status.idle":"2023-07-16T11:57:19.616141Z","shell.execute_reply.started":"2023-07-16T11:57:19.600346Z","shell.execute_reply":"2023-07-16T11:57:19.614968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare modeling ","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\nimport time \n\ntry:\n    from sklearn.metrics import roc_auc_score\n    from sklearn.metrics import f1_score\n    from sklearn.linear_model import Ridge\nexcept:\n    print('Sklearn is not installed')\n\n# Use Keras or tensorflow.keras, don't use both of them.\n# https://stackoverflow.com/questions/63761504/typeerror-the-added-layer-must-be-an-instance-of-class-layer-found-keras-lay\n\n# from tensorflow import keras\n# from tensorflow.keras.models import Sequential\n# from tensorflow.keras.layers import Dense\n# from tensorflow.keras.layers import Dropout\n# from tensorflow.keras.layers import BatchNormalization\n# from tensorflow.keras.metrics import AUC\n\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.layers import Dropout\nfrom keras.layers import BatchNormalization\nfrom keras.metrics import AUC","metadata":{"execution":{"iopub.status.busy":"2023-07-16T11:57:19.617606Z","iopub.execute_input":"2023-07-16T11:57:19.61798Z","iopub.status.idle":"2023-07-16T11:57:19.633789Z","shell.execute_reply.started":"2023-07-16T11:57:19.617945Z","shell.execute_reply":"2023-07-16T11:57:19.632734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get model function","metadata":{}},{"cell_type":"code","source":"def get_model( config_model_features_postprocessing_etc ,  verbose = 100 ):  \n    '''\n    config_model_features_postprocessing_etc - dictionary with configurations\n    '''\n    # See also : https://www.kaggle.com/code/alexandervc/cafa5-mlp-tune-1#Get-model-and-other-auxiliary-functions\n    \n    model_cfg = config_model_features_postprocessing_etc['Model']\n    \n    if verbose >= 100:\n        print( model_cfg )\n    \n    if model_cfg[0] == 'Ridge':\n        alpha = model_cfg[1]['alpha'];\n        str_model_id = 'Ridge'+str(alpha)\n        model = Ridge(alpha=alpha)\n        \n    elif model_cfg[0] == 'KMLP' :\n        \n        model = Sequential(); str_model_id = 'KMLP'\n        layers_sizes = model_cfg[1]['Layers']\n        if 'Dropouts' in model_cfg[1].keys():\n            list_droupouts = model_cfg[1]['Dropouts']\n        else:\n            list_droupouts = []\n        if 'BatchNormalizations' in model_cfg[1].keys():\n            list_batchnormalization = model_cfg[1]['BatchNormalizations']\n        else:\n            list_batchnormalization = []\n            \n        nfeats = X.shape[1]#  model_cfg[1]['input_dim']\n        nlabels = Y.shape[1]#  model_cfg[1]['output_dim']\n        \n        ######### First Layer #################################################################################\n        i_layer = 0 \n        layer_dim = layers_sizes[i_layer]\n        model.add(Dense(layer_dim, activation='relu', input_dim=nfeats))  ; str_model_id += '_L1_'+str(layer_dim) \n        if len( list_batchnormalization ) > i_layer:\n            if list_batchnormalization[i_layer]:\n                model.add(BatchNormalization())\n                str_model_id += '_BN'\n        if (len( list_droupouts ) > i_layer) and ( list_droupouts[i_layer] is not None ):\n            droupout_rate  = list_droupouts[i_layer]\n            model.add(Dropout(droupout_rate))\n            str_model_id += '_DR'+str( np.round(droupout_rate ,2) ) \n            \n        ######### Middle layers ##########################################################################################\n        for i_layer in range(1, len( layers_sizes ) ) : \n            layer_dim = layers_sizes[i_layer]\n            if layer_dim is None: break  \n            model.add(Dense(layer_dim, activation='relu'));                 str_model_id += '_L'+str(i_layer+1)+'_'+str(layer_dim)    \n            if len( list_batchnormalization ) > i_layer:\n                if list_batchnormalization[i_layer]:\n                    model.add(BatchNormalization())\n                    str_model_id += '_BN'\n            if (len( list_droupouts ) > i_layer) and ( list_droupouts[i_layer] is not None ):\n                droupout_rate  = list_droupouts[i_layer]\n                model.add(Dropout(droupout_rate))\n                str_model_id += '_DR'+str( np.round(droupout_rate ,2) ) \n            \n        ######### Last layer ########################################################################################\n        model.add(Dense(nlabels, activation='sigmoid'))\n        \n        model.compile(loss='binary_crossentropy',\n                        optimizer='adam',\n                        metrics=[AUC()])        \n                      \n    if verbose >= 10:\n        print( str_model_id,  model )\n        \n        \n    return model, str_model_id","metadata":{"execution":{"iopub.status.busy":"2023-07-16T11:57:19.635111Z","iopub.execute_input":"2023-07-16T11:57:19.635451Z","iopub.status.idle":"2023-07-16T11:57:19.655407Z","shell.execute_reply.started":"2023-07-16T11:57:19.635422Z","shell.execute_reply":"2023-07-16T11:57:19.654384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fit model function","metadata":{}},{"cell_type":"code","source":"def model_fit(model , X_loc, Y_loc, config_model_features_postprocessing_etc  ):\n    \n    model_cfg = config_model_features_postprocessing_etc['Model']\n    \n    if verbose >= 100:\n        print( model_cfg )\n    \n    dict_prm_for_fit = {t : model_cfg[1][t] for t in ['epochs' ,   'batch_size' , 'verbose']   if t in model_cfg[1].keys()  }\n    \n    # if ('epochs' in model_cfg[1].keys()  ) and (   'batch_size' in model_cfg[1].keys() ) and ('verbose' in  model_cfg[1].keys()  ) :\n    \n    if len( dict_prm_for_fit ) == 0:\n        model.fit( X_loc, Y_loc  )\n    else:\n        model.fit( X_loc, Y_loc , **dict_prm_for_fit  )\n    ","metadata":{"execution":{"iopub.status.busy":"2023-07-16T11:57:19.656935Z","iopub.execute_input":"2023-07-16T11:57:19.657939Z","iopub.status.idle":"2023-07-16T11:57:19.671569Z","shell.execute_reply.started":"2023-07-16T11:57:19.657875Z","shell.execute_reply":"2023-07-16T11:57:19.670412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"%%time\nimport time \n\nn_folds = 2 # How many folds use for CV\n\nn_folds_to_process = 100 # To speed-up we can compute only on 1,2... folds\n\ncfg0 = {'Model': ['Ridge',{'alpha':1000} ] } \ncfg1 = {'Model': ['Ridge',{'alpha':100} ] } \ncfg2 = {'Model': ['Ridge',{'alpha':10} ] } \nlist_main_configs_models_etc = [cfg0,cfg1,cfg2]\n\ncutoff_threshold_low = 0.1 # To speed-up metric calculation (and also avoid RAM crashes) we do not use predictions lower than threshold predictions \n\nverbose = 100_000\n\n\nstr_cv_info = 'cv_folds_'+str(n_folds)\nvec_folds_labels = df_folds['cv_folds_'+str(n_folds) ]\n\nt0all_models_start = time.time();\ndf_foldwise_stat = pd.DataFrame(); \ndn = '/kaggle/input/protein-embeddings-4-cafa5-selected/Embeddings_43k/'\nlist_emb_fn = ['emb_T5_train_float32_cut43k.npy', 'emb_esm2t48B15S5120_train_float32_cut43k.npy', 'emb_esm2t33M650S1280VF_train_float32_cut43k.npy',\n              'emb_esm2t30M150S640_train_embeds_float32_cut43k.npy']\n#/kaggle/input/protein-embeddings-4-cafa5-selected/Embeddings_43k/\nfor fn in list_emb_fn:\n    X = np.load(dn + fn)[:n_samples_to_consider ,:]\n    str_features_id = fn.split('_')[1]\n    print(X.shape, str_features_id , fn)\n\n    for i_cfg, config_model_features_postprocessing_etc in enumerate( list_main_configs_models_etc): # [cfg0, cfg1 ]:\n        str_cfg_count_id = 'C'+str(i_cfg)\n        if verbose >= 100:\n            print( config_model_features_postprocessing_etc )\n    \n        Y_pred_oof_one_model = np.zeros( Y.shape  , dtype = np.float16  )\n        fit_time = 0; predict_time = 0; scoring_time = 0; t0model_start = time.time();\n        ##################################################################\n        ## Loop over folds  \n        ##################################################################\n        for i_fold,uv in enumerate(np.sort(vec_folds_labels.unique())[:n_folds_to_process]):\n            ########## Get model #########################################\n            model, str_model_id = get_model( config_model_features_postprocessing_etc ,  verbose = 10 )\n            \n            str_full_id = str_model_id +' '+  str_features_id + ' ' + str_cfg_count_id\n            \n            \n\n            ########## Get folds #########################################\n            m = vec_folds_labels == uv\n            IX_train = np.where( ~m)[0]\n            IX_test  = np.where(  m)[0]\n            if verbose >= 100:\n                print('Folds info:',i_fold,uv, m.sum() )        \n\n            ########## Fit model #########################################\n            t0 = time.time()\n            # model.fit(X[IX_train,:], Y[IX_train,:] )\n            model_fit(model , X[IX_train,:],Y[IX_train,:], config_model_features_postprocessing_etc  )\n            fit_time_fold = ( time.time() - t0 ); fit_time += fit_time_fold     \n            if verbose >= 100:\n                print('%.1f secs passed on fit'%(fit_time_fold))    \n\n            ########## Predict on local test, OOF  #########################################\n            flag_predict_local_test = True \n            if flag_predict_local_test: \n                IX_loc = IX_test\n                t0 = time.time()\n                Y_pred = model.predict(X[IX_loc,:])\n                time_predict_on_test_fold = time.time() - t0\n                Y_pred_oof_one_model[IX_loc,:] = Y_pred \n                if verbose >= 100:\n                    print('%.1f secs passed on predict local test'%(time_predict_on_test_fold))\n\n            flag_scoring_each_model = True      \n            if flag_scoring_each_model:\n                update_scores(df_foldwise_stat, str_full_id, fold_id = uv,   Y = Y[IX_loc,:] ,  Y_pred = Y_pred,  \n                       IX = IX_loc,      cutoff_threshold_low = cutoff_threshold_low ,               fit_time = fit_time_fold,\n                       predict_time = time_predict_on_test_fold, blend_mode = 'mean', str_cv_info = str_cv_info  )                    \n\n\n\n            ### Optional not really necessary part - predict on train - see gap between train / test score         \n            ### Optional not really necessary part - predict on train - see gap between train / test score         \n            ### Optional not really necessary part - predict on train - see gap between train / test score         \n            ### Optional not really necessary part - predict on train - see gap between train / test score         \n\n            ########## Predict on local TRAIN , OOF  #########################################\n            flag_predict_local_train = True \n            if flag_predict_local_train: \n                IX_loc = IX_train\n                t0 = time.time()\n                Y_pred = model.predict(X[IX_loc,:])\n                time_predict_on_train_fold = time.time() - t0\n                if verbose >= 100:\n                    print('%.1f secs passed on predict local test'%(time_predict_on_train_fold))\n\n\n            flag_scoring_each_model_train =  True      \n            if flag_scoring_each_model_train:\n                update_scores(df_foldwise_stat, str_full_id +' Train', fold_id = uv,   Y = Y[IX_loc,:] ,  Y_pred = Y_pred,  \n                       IX = IX_loc,      cutoff_threshold_low = cutoff_threshold_low ,               fit_time = fit_time_fold,\n                       predict_time = time_predict_on_test_fold, blend_mode = 'mean', str_cv_info = str_cv_info  )                    \n\n        if verbose >= 100:\n            display( df_foldwise_stat.tail(10) )\n\nprint('%.1f secs passed on full modelling'%(time.time() - t0all_models_start  ))    \n\ncol = df_foldwise_stat.columns[2]\ndf_foldwise_stat.sort_values(col, ascending = False)","metadata":{"execution":{"iopub.status.busy":"2023-07-16T11:57:19.67347Z","iopub.execute_input":"2023-07-16T11:57:19.67391Z","iopub.status.idle":"2023-07-16T12:41:03.112985Z","shell.execute_reply.started":"2023-07-16T11:57:19.673859Z","shell.execute_reply":"2023-07-16T12:41:03.111984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_foldwise_stat.to_csv('df_foldwise_models_stat.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-16T12:41:03.114455Z","iopub.execute_input":"2023-07-16T12:41:03.11551Z","iopub.status.idle":"2023-07-16T12:41:03.122373Z","shell.execute_reply.started":"2023-07-16T12:41:03.115474Z","shell.execute_reply":"2023-07-16T12:41:03.121489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Show results","metadata":{}},{"cell_type":"code","source":"col1 = df_foldwise_stat.columns[0]\nd1 = df_foldwise_stat.groupby(col1).mean().sort_values(col, ascending = False).reset_index()\ndisplay(d1)\nd1.to_csv('df_models_stat.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-16T12:41:03.123807Z","iopub.execute_input":"2023-07-16T12:41:03.124249Z","iopub.status.idle":"2023-07-16T12:41:03.162292Z","shell.execute_reply.started":"2023-07-16T12:41:03.124211Z","shell.execute_reply":"2023-07-16T12:41:03.161171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_foldwise_stat.sort_values(col, ascending = False)","metadata":{"execution":{"iopub.status.busy":"2023-07-16T12:41:03.163941Z","iopub.execute_input":"2023-07-16T12:41:03.164418Z","iopub.status.idle":"2023-07-16T12:41:03.190086Z","shell.execute_reply.started":"2023-07-16T12:41:03.164379Z","shell.execute_reply":"2023-07-16T12:41:03.189331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_stat.sort_values(col, ascending = False).to_csv('df_Ridge_RocAuc_models_stat.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-16T12:41:03.191492Z","iopub.execute_input":"2023-07-16T12:41:03.191801Z","iopub.status.idle":"2023-07-16T12:41:03.200494Z","shell.execute_reply.started":"2023-07-16T12:41:03.191772Z","shell.execute_reply":"2023-07-16T12:41:03.199319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )\n","metadata":{"execution":{"iopub.status.busy":"2023-07-16T12:41:03.2045Z","iopub.execute_input":"2023-07-16T12:41:03.205025Z","iopub.status.idle":"2023-07-16T12:41:03.212849Z","shell.execute_reply.started":"2023-07-16T12:41:03.204994Z","shell.execute_reply":"2023-07-16T12:41:03.211957Z"},"trusted":true},"execution_count":null,"outputs":[]}]}