{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Multitarget(binary) Classification with XGBoost","metadata":{}},{"cell_type":"markdown","source":"This notebook applies hyperparameter tuning on XGBoost model and make predictions by applying a softmax output which can be treated as probabilities and then prepare the data in submission format","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport xgboost as xgb\nfrom sklearn.metrics import confusion_matrix, roc_auc_score, accuracy_score, f1_score, roc_curve\nfrom sklearn.experimental import enable_halving_search_cv\nfrom sklearn.model_selection import train_test_split, GridSearchCV, HalvingGridSearchCV\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nfrom copy import deepcopy\nimport time\nfrom sklearn.multioutput import MultiOutputClassifier\n\nmpl.rcParams['figure.dpi'] = 200","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:42:29.175908Z","iopub.execute_input":"2023-07-07T21:42:29.176531Z","iopub.status.idle":"2023-07-07T21:42:30.787532Z","shell.execute_reply.started":"2023-07-07T21:42:29.176495Z","shell.execute_reply":"2023-07-07T21:42:30.786514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf_xgb = xgb.XGBClassifier(objective=\"binary:logistic\", random_state=42, tree_method = \"gpu_hist\")","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:42:30.78938Z","iopub.execute_input":"2023-07-07T21:42:30.789727Z","iopub.status.idle":"2023-07-07T21:42:30.794096Z","shell.execute_reply.started":"2023-07-07T21:42:30.789687Z","shell.execute_reply":"2023-07-07T21:42:30.793383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### laod an older baseline model","metadata":{}},{"cell_type":"code","source":"clf_xgb.load_model(\"/kaggle/input/xgbdata/xgboost-full-2905-5pm.json\")","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:42:30.795243Z","iopub.execute_input":"2023-07-07T21:42:30.795823Z","iopub.status.idle":"2023-07-07T21:42:50.913767Z","shell.execute_reply.started":"2023-07-07T21:42:30.795795Z","shell.execute_reply":"2023-07-07T21:42:50.912833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y = np.load(\"/kaggle/input/xgbdata/Y_1499.npy\")","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:42:50.916582Z","iopub.execute_input":"2023-07-07T21:42:50.91717Z","iopub.status.idle":"2023-07-07T21:43:07.490868Z","shell.execute_reply.started":"2023-07-07T21:42:50.917103Z","shell.execute_reply":"2023-07-07T21:43:07.489883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_to_consider = np.load(\"/kaggle/input/xgbdata/Y_1499_labels.npy\")","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:43:07.492277Z","iopub.execute_input":"2023-07-07T21:43:07.492698Z","iopub.status.idle":"2023-07-07T21:43:07.505788Z","shell.execute_reply.started":"2023-07-07T21:43:07.492651Z","shell.execute_reply":"2023-07-07T21:43:07.504573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fn = '/kaggle/input/t5embeds/test_embeds.npy'","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:43:07.507254Z","iopub.execute_input":"2023-07-07T21:43:07.507563Z","iopub.status.idle":"2023-07-07T21:43:07.511367Z","shell.execute_reply.started":"2023-07-07T21:43:07.50754Z","shell.execute_reply":"2023-07-07T21:43:07.510427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.load(fn)\nprint(X.shape)\nX","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:43:07.512677Z","iopub.execute_input":"2023-07-07T21:43:07.513102Z","iopub.status.idle":"2023-07-07T21:43:16.756486Z","shell.execute_reply.started":"2023-07-07T21:43:07.513077Z","shell.execute_reply":"2023-07-07T21:43:16.755355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## If taking new Predictions, run following 2 cells","metadata":{}},{"cell_type":"code","source":"Y_pred_raw = clf_xgb.predict(X, output_margin = True)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T08:59:45.788671Z","iopub.execute_input":"2023-06-06T08:59:45.78927Z","iopub.status.idle":"2023-06-06T09:03:12.500425Z","shell.execute_reply.started":"2023-06-06T08:59:45.789236Z","shell.execute_reply":"2023-06-06T09:03:12.49974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred_probabilities = 1 / (1 + np.exp(-Y_pred_raw))\nY_pred_probabilities","metadata":{"execution":{"iopub.status.busy":"2023-06-06T09:03:12.501457Z","iopub.execute_input":"2023-06-06T09:03:12.501699Z","iopub.status.idle":"2023-06-06T09:03:13.290367Z","shell.execute_reply.started":"2023-06-06T09:03:12.50168Z","shell.execute_reply":"2023-06-06T09:03:13.289091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save the above predictions","metadata":{}},{"cell_type":"code","source":"np.save(\"xgb-prediction-full-probabilities-06-06\", Y_pred_probabilities)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load previously saved predictions","metadata":{}},{"cell_type":"code","source":"Y_pred_probabilities = np.load('/kaggle/input/cafa-5-submission/xgb-prediction-full-probabilities-06-06.npy')","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:43:16.757737Z","iopub.execute_input":"2023-07-07T21:43:16.758143Z","iopub.status.idle":"2023-07-07T21:43:22.359827Z","shell.execute_reply.started":"2023-07-07T21:43:16.758104Z","shell.execute_reply":"2023-07-07T21:43:22.358581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y_pred_binary = np.load(\"/kaggle/input/predictions-t5-test-embeds/xgb-prediction-test-embeds-full-2905-5pm.json.npy\")","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:43:22.361279Z","iopub.execute_input":"2023-07-07T21:43:22.361683Z","iopub.status.idle":"2023-07-07T21:43:30.583017Z","shell.execute_reply.started":"2023-07-07T21:43:22.361648Z","shell.execute_reply":"2023-07-07T21:43:30.581975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preparing data in submission format","metadata":{}},{"cell_type":"code","source":"%%time \ndf_finalSubmission = pd.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:43:30.586767Z","iopub.execute_input":"2023-07-07T21:43:30.587257Z","iopub.status.idle":"2023-07-07T21:43:30.603527Z","shell.execute_reply.started":"2023-07-07T21:43:30.587214Z","shell.execute_reply":"2023-07-07T21:43:30.602196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/protbert-embeddings-for-cafa5/test_ids.npy'\nvec_test_protein_ids = np.load(fn)\nprint(vec_test_protein_ids.shape)\nvec_test_protein_ids","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:43:30.605148Z","iopub.execute_input":"2023-07-07T21:43:30.605638Z","iopub.status.idle":"2023-07-07T21:43:30.666336Z","shell.execute_reply.started":"2023-07-07T21:43:30.605598Z","shell.execute_reply":"2023-07-07T21:43:30.665323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We cannot form the submission file for all the labels at once as it consumes a lot of RAM, so we only considered 1/4th of the data","metadata":{}},{"cell_type":"code","source":"test_data_batch_sz = 35000","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:43:30.667397Z","iopub.execute_input":"2023-07-07T21:43:30.667676Z","iopub.status.idle":"2023-07-07T21:43:30.672819Z","shell.execute_reply.started":"2023-07-07T21:43:30.667653Z","shell.execute_reply":"2023-07-07T21:43:30.671552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nl = []\nfor k in list(vec_test_protein_ids[:test_data_batch_sz]):\n        l += [ k] * (Y_pred_probabilities.shape[1])\nprint(len(l), l[:20])    \n\ndf_finalSubmission['Protein Id'] = l","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:43:30.674413Z","iopub.execute_input":"2023-07-07T21:43:30.674705Z","iopub.status.idle":"2023-07-07T21:43:47.572655Z","shell.execute_reply.started":"2023-07-07T21:43:30.674681Z","shell.execute_reply":"2023-07-07T21:43:47.571269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_to_consider_repeated = [item for _ in range(test_data_batch_sz) for item in labels_to_consider]\nlabels_to_consider_repeated[:10]","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:43:47.573947Z","iopub.execute_input":"2023-07-07T21:43:47.574302Z","iopub.status.idle":"2023-07-07T21:44:15.67179Z","shell.execute_reply.started":"2023-07-07T21:43:47.574276Z","shell.execute_reply":"2023-07-07T21:44:15.670255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(labels_to_consider_repeated)","metadata":{"execution":{"iopub.status.busy":"2023-07-07T21:44:15.673863Z","iopub.execute_input":"2023-07-07T21:44:15.674627Z","iopub.status.idle":"2023-07-07T21:44:15.68282Z","shell.execute_reply.started":"2023-07-07T21:44:15.674596Z","shell.execute_reply":"2023-07-07T21:44:15.681682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission['GO Term Id'] = labels_to_consider_repeated","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission['Prediction'] = Y_pred_probabilities[:test_data_batch_sz].ravel()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission.to_csv(\"submission-lstm-probert-35000-0207.tsv\",header=False, index=False, sep=\"\\t\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_finalSubmission.info()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \ndf_finalSubmission.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nplt.figure(figsize = (15,4))\nplt.hist(df_finalSubmission['Prediction'].values, bins = 10 )\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(np.unique(Y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-06-04T08:21:35.002502Z","iopub.execute_input":"2023-06-04T08:21:35.002943Z","iopub.status.idle":"2023-06-04T08:21:44.174301Z","shell.execute_reply.started":"2023-06-04T08:21:35.002908Z","shell.execute_reply":"2023-06-04T08:21:44.173006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluating the predictions","metadata":{}},{"cell_type":"code","source":"l = []\nfor i in range(Y.shape[1]):\n    if len(np.unique(Y[IX_test,i]) ) > 1:\n        s = roc_auc_score(Y[IX_test,i], Y_pred[:,i]);\n    else:\n        s = 0.5\n    l.append(s)        \n    if i %10 == 0:\n        print(i, s)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_models_stat = pd.DataFrame()\ndf_models_stat.loc[\"XGB\",'RocAuc Mean Test'] = np.mean(l)\ndf_models_stat.loc[\"XGB\",'Test Size'] = len(IX_test)\ndf_models_stat","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.hist(l)\nplt.show()\npd.Series(l).describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l = []\nfor i in range(Y.shape[1]):\n    if len(np.unique(Y[IX_test,i]) ) > 1:\n        s = accuracy_score(Y[IX_test,i], Y_pred[:,i]);\n    else:\n        s = 0.5\n    l.append(s)        \n    if i %10 == 0:\n        print(i, s)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## HYPERPARAMETER TUNING","metadata":{}},{"cell_type":"code","source":"IX = np.arange(len(X))\nprint(IX.shape)\nprint(IX)\nIX_train, IX_test, _,_ = train_test_split( IX, IX, train_size=0.1, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-06-07T06:31:39.936332Z","iopub.execute_input":"2023-06-07T06:31:39.937235Z","iopub.status.idle":"2023-06-07T06:31:39.953016Z","shell.execute_reply.started":"2023-06-07T06:31:39.937206Z","shell.execute_reply":"2023-06-07T06:31:39.951981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = X[IX_train,:], X[IX_test,:], Y[IX_train,:], Y[IX_test,:]","metadata":{"execution":{"iopub.status.busy":"2023-06-07T06:31:39.954574Z","iopub.execute_input":"2023-06-07T06:31:39.955301Z","iopub.status.idle":"2023-06-07T06:31:42.570868Z","shell.execute_reply.started":"2023-06-07T06:31:39.955267Z","shell.execute_reply":"2023-06-07T06:31:42.569879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Hyperparameter tuning with hyperband search CV.\n#### Code used from the following repo - [https://github.com/thuijskens/scikit-hyperband](http://)","metadata":{}},{"cell_type":"code","source":"import copy\n\nimport numpy as np\nfrom scipy.stats import rankdata\n\nfrom sklearn.utils import check_random_state\nfrom sklearn.model_selection._search import BaseSearchCV, ParameterSampler\n\n\n__all__ = ['HyperbandSearchCV']\n\n\nclass HyperbandSearchCV(GridSearchCV):\n    def __init__(self, estimator, param_distributions,\n                 resource_param='n_estimators', eta=3, min_iter=1,\n                 max_iter=81, skip_last=0, scoring=None, n_jobs=1,\n                 refit=True, cv=None,\n                 verbose=0, pre_dispatch='2*n_jobs', random_state=None,\n                 error_score='raise', return_train_score=False):\n        self.param_distributions = param_distributions\n        self.resource_param = resource_param\n        self.eta = eta\n        self.min_iter = min_iter\n        self.max_iter = max_iter\n        self.skip_last = skip_last\n        self.random_state = random_state\n\n        super(HyperbandSearchCV, self).__init__(\n            estimator=estimator,param_grid=param_distributions, scoring=scoring, n_jobs=n_jobs,\n            refit=refit, cv=cv, verbose=verbose,\n            pre_dispatch=pre_dispatch, error_score=error_score,\n            return_train_score=return_train_score)\n\n    def _run_search(self, evaluate_candidates):\n        self._validate_input()\n\n        s_max = int(np.floor(np.log(self.max_iter / self.min_iter) / np.log(self.eta)))\n        B = (s_max + 1) * self.max_iter\n\n        refit_metric = 'score'\n        random_state = check_random_state(self.random_state)\n\n        if self.skip_last > s_max:\n            raise ValueError('skip_last is higher than the total number of rounds')\n\n        for round_index, s in enumerate(reversed(range(s_max + 1))):\n            n = int(np.ceil(int(B / self.max_iter / (s + 1)) * np.power(self.eta, s)))\n\n            # initial number of iterations per config\n            r = self.max_iter / np.power(self.eta, s)\n            configurations = list(ParameterSampler(param_distributions=self.param_distributions,\n                                                   n_iter=n,\n                                                   random_state=random_state))\n\n            if self.verbose > 0:\n                print('Starting bracket {0} (out of {1}) of hyperband'\n                      .format(round_index + 1, s_max + 1))\n\n            for i in range((s + 1) - self.skip_last):\n\n                n_configs = np.floor(n / np.power(self.eta, i))  # n_i\n                n_iterations = int(r * np.power(self.eta, i))  # r_i\n                n_to_keep = int(np.floor(n_configs / self.eta))\n\n                if self.verbose > 0:\n                    msg = ('Starting successive halving iteration {0} out of'\n                           ' {1}. Fitting {2} configurations, with'\n                           ' resource_param {3} set to {4}')\n\n                    if n_to_keep > 0:\n                        msg += ', and keeping the best {5} configurations.'\n\n                    msg = msg.format(i + 1, s + 1, len(configurations),\n                                     self.resource_param, n_iterations,\n                                     n_to_keep)\n                    print(msg)\n\n                # Set the cost parameter for every configuration\n                parameters = copy.deepcopy(configurations)\n                for configuration in parameters:\n                    configuration[self.resource_param] = n_iterations\n\n                results = evaluate_candidates(parameters)\n\n                if n_to_keep > 0:\n                    top_configurations = [x for _, x in sorted(zip(results['rank_test_%s' % refit_metric],\n                                                                   results['params']),\n                                                               key=lambda x: x[0])]\n\n                    configurations = top_configurations[:n_to_keep]\n\n            if self.skip_last > 0:\n                print('Skipping the last {0} successive halving iterations'\n                      .format(self.skip_last))\n\n    def fit(self, X, y=None, groups=None, **fit_params):\n        \"\"\"Run fit with all sets of parameters.\n\n        Parameters\n        ----------\n        X : array-like, shape = [n_samples, n_features]\n            Training vector, where n_samples is the number of samples and\n            n_features is the number of features.\n\n        y : array-like, shape = [n_samples] or [n_samples, n_output], optional\n            Target relative to X for classification or regression;\n            None for unsupervised learning.\n\n        groups : array-like, with shape (n_samples,), optional\n            Group labels for the samples used while splitting the dataset into\n            train/test set.\nt\n        **fit_params : dict of string -> object\n            Parameters passed to the ``fit`` method of the estimator\n        \"\"\"\n        super().fit(X=X, y=y, groups=groups, **fit_params)\n\n        s_max = int(np.floor(np.log(self.max_iter / self.min_iter) / np.log(self.eta)))\n        B = (s_max + 1) * self.max_iter\n\n        brackets = []\n        for round_index, s in enumerate(reversed(range(s_max + 1))):\n            n = int(np.ceil(int(B / self.max_iter / (s + 1)) * np.power(self.eta, s)))\n            n_configs = int(sum([np.floor(n / np.power(self.eta, i))\n                                 for i in range((s + 1) - self.skip_last)]))\n            bracket = (round_index + 1) * np.ones(n_configs)\n            brackets.append(bracket)\n\n        self.cv_results_['hyperband_bracket'] = np.hstack(brackets)\n\n        return self\n\n    def _validate_input(self):\n        if not isinstance(self.min_iter, int) or self.min_iter <= 0:\n            raise ValueError('min_iter should be a positive integer, got %s' %\n                             self.min_iter)\n\n        if not isinstance(self.max_iter, int) or self.max_iter <= 0:\n            raise ValueError('max_iter should be a positive integer, got %s' %\n                             self.max_iter)\n\n        if self.max_iter < self.min_iter:\n            raise ValueError('max_iter should be bigger than min_iter, got'\n                             'max_iter=%d and min_iter=%d' % (self.max_iter,\n                                                              self.min_iter))\n\n        if not isinstance(self.skip_last, int) or self.skip_last < 0:\n            raise ValueError('skip_last should be an integer, got %s' %\n                             self.skip_last)\n\n        if not isinstance(self.eta, int) or not self.eta > 1:\n            raise ValueError('eta should be a positive integer, got %s' %\n                             self.eta)\n\n        if self.resource_param not in self.estimator.get_params().keys():\n            raise ValueError('resource_param is set to %s, but base_estimator %s '\n                             'does not have a parameter with that name' %\n                             (self.resource_param,\n                              self.estimator.__class__.__name__))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param_grid = {'gamma': [0,6.4,25.6,102.4],\n              'learning_rate': [0.03, 0.3, 1],\n              'max_depth': [3,6,12],\n              'reg_alpha': [0,6.4,25.6],\n              'reg_lambda': [0.5,6.4,25.6]}\n\nsearch0 = HyperbandSearchCV(estimator=clf_xgb,\n                            param_distributions = param_grid,\n                            resource_param='n_estimators',\n                            scoring='roc_auc',\n                           return_train_score=True,\n                           verbose=1,\n                           cv=3)\nsearch0.fit(X_train,y_train)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"search0.best_params_","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The following is hyperparameter tuning with GridSearchCV but it never worked because it was taking way too long","metadata":{}},{"cell_type":"code","source":"results_dict = {}\n\ndefault_params = {}\ngparams = clf_xgb.get_params()\nfor key in gparams.keys():\n    gp = gparams[key]\n    default_params[key] = [gp]\n\n    \n#benchmark model. Grid search is not performed, since only single values are provided as parameter grid.\n#However, cross-validation is still executed\nclf0 = GridSearchCV(estimator=clf_xgb, \n                    scoring='roc_auc', \n                    param_grid=default_params, \n                    return_train_score=True, \n                    verbose=1, \n                    cv=3)\nclf0.fit(X[IX_train,:], Y[IX_train,:])\n\ndf = pd.DataFrame(clf0.cv_results_)\n\n#Taking predections\ntrain_predictions = clf0.predict(X[IX_train,:])\ntest_predictions = clf0.predict(X[IX_test,:])\n\n#Accuracy Score\naccs_train = accuracy_score(Y[IX_train,:], train_predictions)\naccs_test = accuracy_score(Y[IX_test,:], test_predictions)\n\n#confusion matrix\ncfm_train = confusion_matrix(y_train, train_predictions)\ncfm_test = confusion_matrix(y_test, test_predictions)\n\n#F1 score\nf1s_train_p1 = f1_score(Y[IX_train,:], train_predictions, pos_label=1)\nf1s_train_p0 = f1_score(Y[IX_train,:], train_predictions, pos_label=0)\nf1s_test_p1 = f1_score(Y[IX_test,:], test_predictions, pos_label=1)\nf1s_test_p0 = f1_score(Y[IX_test,:], test_predictions, pos_label=0)\n\n#roc auc score\ntest_ras = roc_auc_score(Y[IX_test,:], clf0.predict_proba(X[IX_test,:])[:,1])\n\nbp = clf0.best_params_\n\nresults_dict['xgbc0'] = {'iterable_parameter': np.nan,\n                         'classifier': deepcopy(clf0),\n                         'cv_results': df.copy(),\n                         'cfm_train': cfm_train,\n                         'cfm_test': cfm_test,\n                         'train_accuracy': accs_train,\n                         'test_accuracy': accs_test,\n                         'train F1-score label 1': f1s_train_p1,\n                         'train F1-score label 0': f1s_train_p0,\n                         'test F1-score label 1': f1s_test_p1,\n                         'test F1-score label 0': f1s_test_p0,\n                         'test roc auc score': test_ras,\n                         'best_params': bp}","metadata":{"execution":{"iopub.status.busy":"2023-06-07T06:31:42.572268Z","iopub.execute_input":"2023-06-07T06:31:42.572888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#creating deepcopy of default parameters before manipulations\nparams = deepcopy(default_params)\n\n#setting grid of selected parameters for iteration\nparam_grid = {'gamma': [0,6.4,25.6],\n              'learning_rate': [0.03, 0.3, 1],\n              'max_depth': [3,6,12],\n              'n_estimators': [50,100,150],\n              'reg_alpha': [0,6.4,25.6],\n              'reg_lambda': [0,6.4,25.6]}\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_candidates = 1\n\nfor params in param_grid.values():\n    n_candidates *= len(params)\nn_candidates","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#start time\nt0 = time.time()\n#No. of jobs\ngcvj = np.cumsum([len(x) for x in param_grid.values()])[-1]\n\n#iteration loop. Each selected parameter iterated separately\nfor i,grid_key in enumerate(param_grid.keys()):\n    \n    #variable for measuring iteration time\n    loop_start = time.time()\n       \n    #creating param_grid argument for GridSearchCV:\n    #listing grid values of current iterable parameter and wrapping non-iterable parameter single values in list\n    for param_key in params.keys():\n        if param_key == grid_key:\n            params[param_key] = param_grid[grid_key]\n        else:\n            #use best parameters of last iteration\n            try:\n                param_value = [clf.best_params_[param_key]]\n                params[param_key] = param_value\n            #use benchmark model parameters for first iteration\n            except:\n                param_value = [clf0.best_params_[param_key]]\n                params[param_key] = param_value\n    \n    #classifier instance of current iteration\n    xgbc = xgb.XGBClassifier(**default_params)\n    \n    #GridSearch instance of current iteration\n   \n    clf = MultiOutputClassifier(HalvingGridSearchCV(estimator=xgbc, param_grid=params, scoring='accuracy', return_train_score=True, verbose=1, cv=3, actor=3))\n    clf.fit(X_train, y_train.values.ravel())\n    \n    #results dataframe\n    df = pd.DataFrame(clf.cv_results_)\n    \n    #predictions - inputs to confusion matrix\n    train_predictions = clf.predict(X_train)\n    test_predictions = clf.predict(X_test)\n    \n    #confusion matrices\n    cfm_train = confusion_matrix(y_train, train_predictions)\n    cfm_test = confusion_matrix(y_test, test_predictions)\n    \n    #accuracy scores\n    accs_train = accuracy_score(y_train, train_predictions)\n    accs_test = accuracy_score(y_test, test_predictions)\n    \n    #F1 scores for each train/test label\n    f1s_train_p1 = f1_score(y_train, train_predictions, pos_label=1)\n    f1s_train_p0 = f1_score(y_train, train_predictions, pos_label=0)\n    f1s_test_p1 = f1_score(y_test, test_predictions, pos_label=1)\n    f1s_test_p0 = f1_score(y_test, test_predictions, pos_label=0)\n    \n    #Area Under the Receiver Operating Characteristic Curve\n    test_ras = roc_auc_score(y_test, clf.predict_proba(X_test)[:,1])\n    \n    #best parameters\n    bp = clf.best_params_\n    \n    #storing computed values in results dictionary\n    results_dict[f'xgbc{i+1}'] = {'iterable_parameter': grid_key,\n                                  'classifier': deepcopy(clf),\n                                  'cv_results': df.copy(),\n                                  'cfm_train': cfm_train,\n                                  'cfm_test': cfm_test,\n                                  'train_accuracy': accs_train,\n                                  'test_accuracy': accs_test,\n                                  'train F1-score label 1': f1s_train_p1,\n                                  'train F1-score label 0': f1s_train_p0,\n                                  'test F1-score label 1': f1s_test_p1,\n                                  'test F1-score label 0': f1s_test_p0,\n                                  'test roc auc score': test_ras,\n                                  'best_params': bp}\n    \n    #variable for measuring iteration time\n    elapsed_time = time.time() - loop_start\n    print(f'iteration #{i+1} finished in: {elapsed_time} seconds')\n\n#stop time\nt1 = time.time()\n\n#elapsed time\ngcvt = t1 - t0","metadata":{"execution":{"iopub.status.busy":"2023-06-07T03:47:44.858935Z","iopub.status.idle":"2023-06-07T03:47:44.859407Z","shell.execute_reply.started":"2023-06-07T03:47:44.859158Z","shell.execute_reply":"2023-06-07T03:47:44.85918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#number of rows depend on number of iterations\nnrows = len(results_dict.keys())\n\n#standard group names for confusion matrices\ngroup_names = ['True Neg','False Pos','False Neg','True Pos']\n\n#creating figure\nf, axes = plt.subplots(nrows,2,figsize=(18,8*nrows));\n\n#iteratively plotting train/test accuracy scores and test confusion matrix\nfor i,ax in enumerate(axes):\n    \n    #current key of results dictionary\n    ckey = list(results_dict.keys())[i] \n    \n    #plotting scores for models other than the benchark model\n    if ckey != 'xgbc0':\n        x1 = results_dict[ckey]['cv_results'].loc[:,'mean_train_score']\n        x2 = results_dict[ckey]['cv_results'].loc[:,'mean_test_score']\n        \n        ax[0].plot(x1, label='train scores', color='blue');\n        ax[0].plot(x2, label='test scores', color='red');\n        ax[0].set_title(f'Iteration #{i+1} results');\n               \n        ax[0].set_xticks(list(range(0,len([x[results_dict[ckey]['iterable_parameter']] for x in results_dict[ckey]['cv_results']['params']]))));\n        ax[0].set_xticklabels(sorted([x[results_dict[ckey]['iterable_parameter']] for x in results_dict[ckey]['cv_results']['params']]));\n    \n        ax[0].grid('major');\n        ax[0].legend();\n        ax[0].set_xlabel(results_dict[ckey]['iterable_parameter'])\n        ax[0].set_ylabel('mean score');\n    \n    #leaving scores plot blank for benchmark model\n    else:\n        ax[0].axis('off')\n        ax[0].text(x=0.5, y=0.5, s='No iteration has been performed', fontsize=16, va='center', ha='center')\n    \n    #computing variables for specific confusion matrix\n    group_counts = [\"{0:0.0f}\".format(value) for value in results_dict[ckey]['cfm_test'].flatten()]\n    group_percentages = [\"{0:.2%}\".format(value) for value in results_dict[ckey]['cfm_test'].flatten()/np.sum(results_dict[ckey]['cfm_test'])]\n    labels = [f\"{v1}\\n{v2}\\n{v3}\" for v1, v2, v3 in zip(group_names,group_counts,group_percentages)]\n    labels = np.asarray(labels).reshape(2,2)\n    \n    #plotting confusion matrix\n    sns.heatmap(results_dict[ckey]['cfm_test'], annot=labels, fmt='', cmap='Blues', ax=ax[1])\n    \nplt.show();","metadata":{"execution":{"iopub.status.busy":"2023-06-06T23:51:52.279896Z","iopub.status.idle":"2023-06-06T23:51:52.280585Z","shell.execute_reply.started":"2023-06-06T23:51:52.280328Z","shell.execute_reply":"2023-06-06T23:51:52.280349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}