{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":84896,"databundleVersionId":10305135},{"sourceType":"datasetVersion","sourceId":9178166,"datasetId":5547076,"databundleVersionId":9360233},{"sourceType":"datasetVersion","sourceId":10118268,"datasetId":6215722,"databundleVersionId":10400356}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import","metadata":{"execution":{"iopub.status.busy":"2024-11-05T05:46:04.060015Z","iopub.execute_input":"2024-11-05T05:46:04.060509Z","iopub.status.idle":"2024-11-05T05:47:05.864174Z","shell.execute_reply.started":"2024-11-05T05:46:04.060451Z","shell.execute_reply":"2024-11-05T05:47:05.862797Z"}}},{"cell_type":"code","source":"from IPython.display import clear_output","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:15:39.034929Z","iopub.execute_input":"2024-12-06T23:15:39.035504Z","iopub.status.idle":"2024-12-06T23:15:39.070786Z","shell.execute_reply.started":"2024-12-06T23:15:39.035448Z","shell.execute_reply":"2024-12-06T23:15:39.069445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"AUTOGLUON = False\n# AUTOGLUON =  True\n# !pip install ray==2.10.0 autogluon.tabular ipywidgets catboost==1.2.5\n# clear_output()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:15:39.072794Z","iopub.execute_input":"2024-12-06T23:15:39.073122Z","iopub.status.idle":"2024-12-06T23:15:39.079282Z","shell.execute_reply.started":"2024-12-06T23:15:39.073091Z","shell.execute_reply":"2024-12-06T23:15:39.0779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport math\n!pip install -q scikit-learn==1.5.2\nclear_output()","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:15:39.108026Z","iopub.execute_input":"2024-12-06T23:15:39.108421Z","iopub.status.idle":"2024-12-06T23:15:57.310738Z","shell.execute_reply.started":"2024-12-06T23:15:39.108386Z","shell.execute_reply":"2024-12-06T23:15:57.309315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import datetime\nimport sys\n\nimport matplotlib\nimport matplotlib as mpl\nimport matplotlib.cm as cmap\nimport matplotlib.colors as mpl_colors\nimport matplotlib.pyplot as plt\nimport matplotlib.ticker as ticker\n\nimport seaborn as sns\n\ndef hex_to_rgb(h):\n    h = h.lstrip('#')\n    return tuple(int(h[i:i+2], 16)/255 for i in (0, 2, 4))\n\n# palette = ['#b4d2b1', '#568f8b', '#1d4a60', '#cd7e59', '#ddb247', '#d15252']\n# palette_rgb = [hex_to_rgb(x) for x in palette]\n# cmap = mpl_colors.ListedColormap(palette_rgb)\n# colors = cmap.colors\nbg_color= '#fdfcf6'\n\nblack, red, green, blue = ['#000000', '#ff0000', '#00ff00', '#0000ff']\n\ncustom_params = {\n    \"axes.spines.right\": False,\n    \"axes.spines.top\": False,\n    'grid.alpha':0.3,\n    'figure.figsize': (16, 6),\n    'axes.titlesize': 'Large',\n    'axes.labelsize': 'Large',\n    'figure.facecolor': bg_color,\n    'axes.facecolor': bg_color\n}\n\nsns.set_theme(\n    style='whitegrid',\n#     palette=sns.color_palette(palette),\n    rc=custom_params\n)\n\nfrom plotly.offline import init_notebook_mode, iplot, plot\nimport plotly.express as px\nimport plotly as py\n#init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\n\nimport scipy.stats as st\n\nfrom warnings import simplefilter\nsimplefilter(\"ignore\")\n\nimport random\nimport os\n\nSEED = 2024\ndef seed_everything(seed=42):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\nseed_everything(SEED)\n\nfrom IPython.display import clear_output\nfrom tqdm import tqdm, trange\n\nfrom sklearn.linear_model import *\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.neural_network import MLPRegressor\n\nfrom sklearn.ensemble import AdaBoostRegressor\nfrom sklearn.ensemble import BaggingRegressor\nfrom sklearn.ensemble import ExtraTreesRegressor\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.ensemble import RandomForestRegressor\n\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor, Pool\nfrom lightgbm import LGBMRegressor\nfrom lightgbm import early_stopping, log_evaluation\n\n\nfrom lightgbm import LGBMClassifier, LGBMRegressor\nfrom lightgbm import early_stopping, log_evaluation\nfrom sklearn.linear_model import LogisticRegression\nimport catboost\nfrom xgboost import XGBClassifier\n\nfrom sklearn.pipeline import make_pipeline, Pipeline\n\n# Encoders\nfrom sklearn.preprocessing import *\nfrom category_encoders.leave_one_out import LeaveOneOutEncoder \nfrom category_encoders import TargetEncoder, WOEEncoder\n\n# Scalers\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, MaxAbsScaler, RobustScaler, Normalizer\n\nfrom sklearn.ensemble import *\nfrom sklearn.compose import *\n\nfrom scipy.stats.mstats import gmean, hmean\nfrom scipy.stats import mode\nfrom numpy import mean, median\n\nimport re\n\nfrom sklearn.model_selection import *\nfrom sklearn.metrics import *\nfrom sklearn.base import clone\nfrom sklearn.calibration import CalibrationDisplay, CalibratedClassifierCV\nfrom sklearn.feature_selection import *\nfrom sklearn.metrics import root_mean_squared_log_error\n\n# if not AUTOGLUON :\n#     import eli5\n#     from eli5.sklearn import PermutationImportance\n#     import shap\n\nfrom termcolor import colored\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, models, Sequential\nfrom tensorflow.keras import backend as K\n\nfrom sklearn.inspection import PartialDependenceDisplay","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:15:57.313745Z","iopub.execute_input":"2024-12-06T23:15:57.314701Z","iopub.status.idle":"2024-12-06T23:16:14.555662Z","shell.execute_reply.started":"2024-12-06T23:15:57.314649Z","shell.execute_reply":"2024-12-06T23:16:14.554515Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# &#129300; PROBLEM","metadata":{}},{"cell_type":"code","source":"PROBLEM = 'regression'\n\nTARGET = 'Premium Amount'\n\ndef load_datasets():\n    train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\n    test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\n    sample_sub = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\n    original = pd.read_csv('/kaggle/input/insurance-premium-prediction/Insurance Premium Prediction Dataset.csv')\n    \n    train = train.drop(['id'], axis=1)\n    test = test.drop(['id'], axis=1)\n\n    train['nans'] = train.isnull().sum(axis=1).astype(float)\n    test['nans'] = test.isnull().sum(axis=1)\n    original['nans'] = original.isnull().sum(axis=1)\n\n    return train, test, original, sample_sub # don't use original\n\ndef SCORE(y_true, y_pred):\n    return root_mean_squared_log_error(y_true, y_pred)\n\ndef LOSS(y_true, y_pred):\n    return root_mean_squared_log_error(y_true, np.clip(y_pred, 20, 4999))\n    \n\nSCORE_NAME = 'RMSLE'\n\nOBJ = -1 # 1 maximize score, -1 minimize score","metadata":{"_kg_hide-input":false,"scrolled":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:16:14.557122Z","iopub.execute_input":"2024-12-06T23:16:14.557891Z","iopub.status.idle":"2024-12-06T23:16:14.565965Z","shell.execute_reply.started":"2024-12-06T23:16:14.557854Z","shell.execute_reply":"2024-12-06T23:16:14.564733Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# &#128297; ENGINE","metadata":{}},{"cell_type":"markdown","source":"## &#128190; Database","metadata":{}},{"cell_type":"code","source":"import shutil\n\nINPUT = '/kaggle/input/insurance-4-12/'\n\nclass DatasetCollection():\n    '''\n        Save and load datasets, i.e. (train, test)\n    '''\n    \n    def __init__(self, dir: str = 'datasets') -> None :\n        self.dir_ = dir \n        if not os.path.exists(self.dir_) : \n            os.mkdir(self.dir_)\n\n        self.data_ = {}\n        \n    def __str__(self) -> str :\n        s = ''\n        for name, ds in self.data_.items() :\n            s += f\"{name}, {ds['description']} :\\n\"\n            for df in ['train', 'test'] :\n                if ds[df] is not None:\n                    s += f'{df}, {ds[df].shape[0]} rows {ds[df].shape[1]} columns\\n'\n            s += '\\n'\n        return s\n    \n    def __repr__(self) -> str :\n        return 'DatasetCollection :\\n\\n' + self.__str__()\n        \n    def load(self) -> None :\n        path = os.path.join(INPUT, self.dir_)\n        if not os.path.exists(path) :\n            print('no datasets found')\n            return\n        listdir = os.listdir(path)\n        for name in listdir :\n            self.data_[name] = {\n                'train' : pd.read_csv(os.path.join(path, name, 'train.csv')) if os.path.exists(os.path.join(path, name, 'train.csv')) else None,\n                'test' : pd.read_csv(os.path.join(path, name, 'test.csv')) if os.path.exists(os.path.join(path, name, 'test.csv')) else None,\n            }\n            if os.path.exists(os.path.join(path, name, 'description.txt')):\n                f = open(os.path.join(path, name, 'description.txt'), 'r')\n                self.data_[name]['description'] = f.read()\n                f.close()\n            else:\n                self.data_[name]['description'] = ''\n                \n    def save(self) -> None :\n        for name, ds in self.data_.items() :\n            if not os.path.exists(os.path.join(self.dir_, name)) : \n                os.mkdir(os.path.join(self.dir_, name))\n            df = self.data_[name]['train']\n            if df is not None :\n                df.to_csv(os.path.join(self.dir_, name, 'train.csv'), index=False)\n            df = self.data_[name]['test']\n            if df is not None :\n                df.to_csv(os.path.join(self.dir_, name, 'test.csv'), index=False)\n            f = open(os.path.join(self.dir_, name, 'description.txt'), 'w')\n            f.write(self.data_[name]['description'])\n            f.close()\n        \n    def put(self, name: str = 'loaded', train: pd.DataFrame = None, test: pd.DataFrame = None, description: str = '',) -> None :\n        self.data_[name] = {\n            'train' : train.copy() if train is not None else None,\n            'test'  : test.copy() if test is not None else None,\n            'description' : description,\n        }\n    \n    def train(self, name: str = 'loaded') -> pd.DataFrame :\n        df = self.data_[name]['train']\n        if df is None :\n            return None\n        return df.copy()\n    \n    def test(self, name: str = 'loaded') -> pd.DataFrame :\n        df = self.data_[name]['test']\n        if df is None :\n            return None\n        return df.copy()\n    \n    def X_test(self, name: str = 'loaded') -> pd.DataFrame :\n        df = self.data_[name]['test']\n        if df is None :\n            return None\n        return df.copy()\n    \n    def X(self, name: str = 'loaded') -> pd.DataFrame :\n        df = self.data_[name]['train'].copy()\n        if df is None :\n            return None\n        cols = df.columns.tolist()\n        if TARGET in cols:\n            cols.remove(TARGET)\n        return df[cols].copy()\n    \n    def y(self, name: str = 'loaded') -> pd.Series :\n        df = self.data_[name]['train'].copy()\n        if df is None :\n            return None\n        cols = df.columns.tolist()\n        if TARGET in cols:\n            return df[TARGET]\n        return None\n\n    def union(self, name: str = 'loaded') -> pd.DataFrame :\n        return pd.concat([self.X(name), self.test(name)])\n    \n    def names(self) -> list :\n        return list(self.data_.keys())\n\n    def summary(self, name : str = 'loaded', dataframes : list = ['train', 'test'], tab : bool = True, plots : bool = False, info : bool = True, nrows : int = 3) -> None :\n        for df_name in dataframes :\n            df = self.data_[name][df_name]\n            print(colored(f'\\n---------- {name} {df_name} ----------:\\n', 'red'))\n                    \n            if tab:\n                display(df.head(nrows))\n                display(df.tail(nrows))\n                print(colored(f'{name} has {df.shape[0]} rows, {df.shape[1]} columns\\n', 'blue'))\n        \n            inf = pd.DataFrame(df.dtypes).reset_index().rename(columns={'index':'column', 0:'type'})\n            df_missed = pd.DataFrame(df.isnull().sum()).reset_index().rename(columns={'index':'column', 0:'missed'})\n            df_unique = pd.DataFrame(df.nunique()).reset_index().rename(columns={'index':'column', 0:'unique'})\n            inf['missed'] = df_missed['missed']\n            inf['unique'] = df_unique['unique']\n            inf['duplicate'] = df.duplicated().sum()\n            \n            desc = pd.DataFrame(df.describe(include='all').transpose())\n            if 'min' in desc.columns.tolist():\n                inf['min'] = desc['min'].values\n                inf['max'] = desc['max'].values\n                inf['avg'] = desc['mean'].values\n                inf['std dev'] = desc['std'].values\n            if 'top' in desc.columns.tolist():\n                inf['top value'] = desc['top'].values\n                inf['Freq'] = desc['freq'].values    \n            \n            if info:\n                display(inf.style.background_gradient(subset='missed', cmap='Reds').background_gradient(subset='unique', cmap='Greens'))\n          \n            if plots:\n                print()\n                if df_missed['missed'].sum() > 0:\n                    fig, ax = plt.subplots(1, 1, figsize=(24, 5))\n                    sns.barplot(df_missed[df_missed['missed'] > 0], x='column', y='missed', ax=ax)\n                    ax.set_title(f'{name} missed values') \n                    ax.bar_label(ax.containers[0])\n                    plt.tight_layout()\n                    plt.show()\n        \n                fig, ax = plt.subplots(1, 1, figsize=(24, 5))\n                sns.barplot(df_unique[df_unique['unique'] > 0], x='column', y='unique', ax=ax)\n                ax.set_title(f'{name} unique values')\n                ax.bar_label(ax.containers[0])\n                plt.tight_layout()\n                plt.show()\n\n\n\nimport json\nnames = {}\n\nclass CrossValidation() :\n\n    def __init__(\n        self,\n        name : str               = 'cv',\n        estimator_class  : str   = 'unknown',\n        dataset_name : str       = 'unknown',\n        params : dict            = {},\n        n_splits : int           = 5, \n        n_repeats : int          = 1, \n        oof_train                = None, \n        oof_true                 = None, \n        oof_test                 = None, \n        submission               = None,\n        oof_scores : list        = [],\n        description : str        = '',\n        iteration_time : float   = 0.0,\n    ) -> None :\n\n        self.summary = {\n            'name' : name,\n            'estimator_class' : estimator_class,\n            'dataset_name' : dataset_name,\n            'params' : params,\n            'n_splits' : n_splits,\n            'n_repeats' : n_repeats,\n            'description' : description,\n            'scores' : oof_scores,\n        }\n        self.oof = {\n            'train'        : oof_train,\n            'test'         : oof_test,\n            'true'         : oof_true,\n        }\n\n    def __str__(self) -> str :\n        s = ''\n        for k, v in self.summary.items() :\n            if k != 'params':\n                s += '    ' + k + ' '* (20-len(k)) + ':' + str(v) + '\\n'\n        s += 'params:\\n'\n        p = self.summary['params']\n        if type(p) is str :\n             s += p + '\\n\\n'\n        else :\n            for k, v in p.items() :\n                s += '    ' + k + ' '* (20-len(k)) + ':' + str(v) + '\\n'\n        for df_name, df in self.oof.items() :\n            if df is not None:\n                s += f'oof {df_name}, {df.shape[0]} rows\\n'\n        s += '\\n'\n        return s\n    \n    def __repr__(self) -> str :\n        return 'CrossValidation:\\n' + self.__str__()\n\n    def load(self, dir : str) -> bool :\n        if not os.path.exists(os.path.join(dir, 'summary.json')):\n            print('directory', dir, 'has no summary')\n            return False\n\n        f = open(os.path.join(dir, 'summary.json'), 'r')\n        summ = f.read()\n        f.close()\n        self.summary = json.loads(summ)\n        \n        for df_name, df in self.oof.items() :\n            self.oof[df_name] = pd.read_csv(os.path.join(dir, f'{df_name}.csv')).to_numpy().flatten() if os.path.exists(os.path.join(dir, f'{df_name}.csv')) else None\n        return True\n\n    def save(self, dir : str) -> None :\n        name = self.summary['name']\n        if not os.path.exists(os.path.join(dir, name)) : \n            os.mkdir(os.path.join(dir, name))\n\n        for df_name, arr in self.oof.items() :\n            if arr is not None:\n                df = pd.DataFrame()\n                df['value'] = arr\n                df.to_csv(os.path.join(dir, name, f'{df_name}.csv'), index=False)\n\n        summary = json.dumps(self.summary, indent=4)\n        f = open(os.path.join(dir, name, 'summary.json'), 'w')\n        f.write(summary)\n        f.close()\n\n    def display(self, estimator, verbose = 2) -> None :\n        estimator.upload_cv(self)\n        estimator.display_cv_results()\n        if verbose > 1 :\n            estimator.display_cv_plots()\n\n    def run(self, estimator, db) :\n        '''\n            rerun cv, put new cv into db\n            return new cv\n        '''\n        s = self.summary\n        cv = estimator.crossvalidate(\n            db.datasets.train(s['dataset_name']),\n            db.datasets.test(s['dataset_name']),\n            name = s['name'],\n            description = s['description'],\n            n_splits = s['n_splits'],\n            n_repeats = s['n_repeats'],\n            dataset_name = s['dataset_name'],\n        )\n        db.cvs.put(cv)\n        return cv\n\n    def submit(self) :\n        '''\n            oof test -> .csv\n        '''\n        sub = sample_sub.copy()\n        sub[TARGET] = self.oof['test_pred']\n        sub.to_csv(f\"cv_{self.summary['name']}.csv\", index=False)\n        display(sub.head(3))\n\n\n\nclass CVCollection():\n    '''\n        Save and load CV settings, i.e. model, perameters, number of\n        splits and repeats, and OOF data.\n    '''\n    \n    def __init__(self, dir: str = 'cvs') -> None :\n        self.dir_ = dir \n        if not os.path.exists(self.dir_) : \n            os.mkdir(self.dir_)\n\n        self.data_ = {}\n        \n    def __str__(self) -> str :\n        s = ''\n        for _, cv in self.data_.items() :\n            s += str(cv) + '\\n'\n        return s\n    \n    def __repr__(self) -> str :\n        return 'CVCollection :\\n\\n' + self.__str__()\n\n    def load(self) -> None :\n        '''\n            load cvs from input directory\n        '''\n        path = os.path.join(INPUT, self.dir_)\n        if not os.path.exists(path) :\n            print('no datasets found')\n            return\n        listdir = os.listdir(path)\n        for name in listdir :\n            cv = CrossValidation()\n            ok = cv.load(os.path.join(path, name))\n            if ok :\n                self.data_[name] = cv\n                \n    def save(self) -> None :\n        '''\n             save all cvs\n        '''\n        for name, cv in self.data_.items() :\n            cv.save(os.path.join(self.dir_))\n        \n    def put(self, cv : CrossValidation) -> None :\n        '''\n            put cv into database\n        '''\n        self.data_[cv.summary['name']] = cv\n\n    def get(self, cv_name : str) -> CrossValidation :\n        '''\n            returns cv by name\n        '''\n        if not cv_name in self.data_:\n            print(colored(f'cv {cv_name} not found'), 'red')\n            return None\n        return self.data_[cv_name]\n\nclass Database():\n    '''\n        Save database when notebook saved with commit,\n        load on run.\n    '''\n    \n    def __init__(self, dir: str = 'database') -> None :\n        self.dir_ = dir \n        if not os.path.exists(self.dir_) : \n            os.mkdir(self.dir_)\n       \n        self.datasets = DatasetCollection(dir = os.path.join(self.dir_, 'datasets'))\n        self.cvs = CVCollection(dir = os.path.join(self.dir_, 'cvs'))\n        \n    def __str__(self) -> str :\n        s = '\\nDatasets:\\n' + self.datasets.__str__()\n        s += '\\nCVs:\\n' + self.cvs.__str__()\n        return s\n    \n    def __repr__(self) -> str :\n        return 'Database :\\n\\n' + self.__str__()\n        \n    def load(self) -> None :\n        self.datasets.load()\n        self.cvs.load()\n        \n    def save(self) -> None :\n        self.datasets.save()\n        self.cvs.save()        \n        shutil.make_archive(self.dir_, 'zip', os.path.join('/kaggle/working', self.dir_))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:16:14.569729Z","iopub.execute_input":"2024-12-06T23:16:14.570138Z","iopub.status.idle":"2024-12-06T23:16:14.633871Z","shell.execute_reply.started":"2024-12-06T23:16:14.570103Z","shell.execute_reply":"2024-12-06T23:16:14.632632Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## ⚖ Averager\nWe will use it to average different prediction in fold with weights optimized ","metadata":{}},{"cell_type":"code","source":"from functools import partial\nimport scipy as sp\n\nclass Averager(object):\n\n    def __init__(self, method='nelder-mead', round_avg=False, options={}):\n        self.weights_ = []\n        self.opt_ = ''\n        self.method_ = method\n        self.round_avg_ = round_avg\n        self.options_ = options\n\n    def _weighted_average(self, weights, values):\n        qty = len(values)\n        sum_values = values[0] * weights[0]\n        sum_weights = weights[0]\n        for i in range(1, qty):\n            sum_values += values[i] * weights[i]\n            sum_weights += weights[i]\n        if self.round_avg_:\n            return int(np.round(sum_values / sum_weights, 0))\n        return sum_values / sum_weights\n\n    def _score(self, weights, values, true_labels):\n        preds = self._weighted_average(weights, values)\n        return LOSS(true_labels, preds)\n\n    def fit(self, values, true_labels):\n        qty = len(values)\n        initial_weights = [1 for _ in range(qty)]\n        score_partial = partial(self._score, values=values, true_labels=true_labels)\n        self.opt_ = sp.optimize.minimize(score_partial, initial_weights, method=self.method_, options=self.options_)\n        self.weights_ = self.opt_['x']\n\n    def predict(self, values):\n        assert len(self.weights_) == len(values), 'Averager error, must be fitted before predict'\n        return self._weighted_average(self.weights_, values)\n\n    def fit_predict(self, values, true_labels):\n        self.fit(values, true_labels)\n        return self.predict(values)\n\n    def weights(self):\n        return self.weights_\n\n    def optimization(self):\n        return self.opt_\n\n","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:16:14.635315Z","iopub.execute_input":"2024-12-06T23:16:14.635808Z","iopub.status.idle":"2024-12-06T23:16:14.652521Z","shell.execute_reply.started":"2024-12-06T23:16:14.635762Z","shell.execute_reply":"2024-12-06T23:16:14.651276Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## &#128188; Estimators Classes","metadata":{}},{"cell_type":"code","source":"class Estimator():\n\n    def __init__(self, name : str = 'model', params : list = {}, verbose : int = 1) :\n        self.name = name\n        self.models_ = []\n        self.model_ = None\n        self.params_ = params\n        self.verbose_ = verbose\n        \n        self.cv = None\n\n    def fit(self, X, y):\n        pass\n\n    def fit_predict(self, X, y, X_val, y_val):\n        pass\n\n    def fit_predict_proba(self, X, y, X_val, y_val):\n        pass\n\n    def predict_proba(self, X):\n        pass\n\n    def predict(self, X):\n        pass\n\n    def crossvalidate(\n            train_: pd.DataFrame, \n            test_:  pd.DataFrame,\n            n_splits=5, n_repeats=1, random_state=42, verbose=1, dataset_name='unknown', use_tqdm=True, clear=True,\n        ) -> CrossValidation :\n        pass\n    \n    def display_cv_plots():\n        pass\n    \n    def display_cv_results():\n        pass\n    \n    def upload_cv(self, cv : CrossValidation) -> None :\n        self.cv = cv\n    \n    def submit():\n        pass\n  ","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:16:14.654092Z","iopub.execute_input":"2024-12-06T23:16:14.65508Z","iopub.status.idle":"2024-12-06T23:16:14.669579Z","shell.execute_reply.started":"2024-12-06T23:16:14.655014Z","shell.execute_reply":"2024-12-06T23:16:14.668444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Regressor(Estimator):\n\n    def fit_predict(self, X, y, X_val, y_val):\n        self.fit(X, y, X_val, y_val)\n        return self.predict(X_val)\n\n    def fit_predict_proba(self, X, y, X_val, y_val):\n        self.fit(X, y, X_val, y_val)\n        return self.predict(X_val)\n\n    def predict_proba(self, X):\n        return self.predict(X)\n    \n    def predict(self, X):\n        assert self.model_ is not None, 'Model error, must be fitted before predict'\n        return self.model_.predict(X)\n        \n    def crossvalidate(self,\n            train_: pd.DataFrame, \n            test_:  pd.DataFrame,\n            name : str = None,            \n            description : str = '',\n            n_splits=5, n_repeats=1, random_state=42, verbose=2, dataset_name='unknown', use_tqdm=True, clear=True,\n        ) -> CrossValidation :\n\n        if name is None :\n            name = self.name\n        \n        # debug\n        train, test = train_.copy(), test_.copy()\n        train_n_rows = train.shape[0]\n        if verbose > 0 :\n            print(f'\\n---------- cv: train: {train.shape}, test: {test.shape} -----------\\n')\n\n        features = test.columns.to_list()\n\n        folds = RepeatedKFold(n_splits=n_splits, n_repeats=n_repeats, random_state=random_state)\n\n        oof_test = np.zeros(len(test))\n        oof_train = np.zeros(len(train))\n        oof_true = np.zeros(len(train))\n        oof_scores = []\n\n        assert train_n_rows == len(oof_train) and train_n_rows == len(oof_true), f\"train ({train_n_lows}), oof_test_proba ({len(oof_train_proba)}), oof_true ({len(oof_true)}) doesn't match\"\n        \n        if use_tqdm and verbose > 0:\n            data = tqdm(enumerate(folds.split(train[features], train[TARGET])), desc='Fold', total=n_splits*n_repeats, file=sys.stdout, colour='GREEN')\n        else:\n            data = enumerate(folds.split(train[features], train[TARGET]))\n\n        start = datetime.datetime.now()\n\n        for i, (train_idx,val_idx) in data:\n\n            train_labels =  train.loc[train_idx, TARGET]\n            val_labels =  train.loc[val_idx, TARGET]\n            train_features = train.loc[train_idx, features]\n            val_features = train.loc[val_idx, features]\n\n            train_labels_log = np.log1p(train_labels)\n            val_labels_log = np.log1p(val_labels)\n    \n            val_preds_log =  self.fit_predict(train_features, train_labels_log, val_features, val_labels_log)\n            test_preds_log = self.predict(test)\n\n            val_preds = np.clip(np.expm1(val_preds_log), a_min = 20.0, a_max = 4999.0)\n            test_preds = np.clip(np.expm1(test_preds_log), a_min = 20.0, a_max = 4999.0)\n\n            oof_test += test_preds\n\n            score = SCORE(val_labels, val_preds)\n            oof_scores.append(score)\n            \n            oof_train[val_idx] = val_preds\n            oof_true[val_idx] = val_labels\n            assert train_n_rows == len(oof_train) and train_n_rows == len(oof_true), f\"train ({train_n_lows}), oof_test_proba ({len(oof_train_proba)}), oof_true ({len(oof_true)}) doesn't match\"\n\n            if clear:\n                clear_output(wait=True)\n \n            if verbose > 0:\n                print('\\nfold', i, SCORE_NAME, score, '\\n')\n\n        if clear:\n            clear_output(wait=True)\n\n        iteration_time = (datetime.datetime.now() - start).total_seconds() / n_splits * n_repeats\n\n        oof_test /= n_splits * n_repeats\n\n        assert train_n_rows == len(oof_train) and train_n_rows == len(oof_true), f\"train ({train_n_lows}), oof_test_proba ({len(oof_train_proba)}), oof_true ({len(oof_true)}) doesn't match\"\n        \n        self.cv = CrossValidation(\n            name                = name,\n            estimator_class     = self.name,\n            dataset_name        = dataset_name,\n            params              = self.params_,\n            n_splits            = n_splits, \n            n_repeats           = n_repeats, \n            oof_train           = oof_train, \n            oof_true            = oof_true, \n            oof_test            = oof_test, \n            oof_scores          = oof_scores,\n            description         = description,\n            iteration_time      = iteration_time,            \n        )\n\n        if verbose > 0:\n            self.display_cv_results()\n        if verbose > 1:\n            self.display_cv_plots()\n        return self.cv\n    \n    def cv_scores(self):       \n        oof_train = self.cv.oof['train']\n        oof_true = self.cv.oof['true']\n\n        if 'scores' in self.cv.summary :\n            mean_oof_score = np.mean(self.cv.summary['scores'])\n        else:\n            mean_oof_score = None\n            \n        score = SCORE(oof_true, oof_train)\n        R2 = r2_score(oof_true, oof_train)\n        \n        self.cv_scores_ = {\n            'Model': self.name,\n            'Dataset': self.cv.summary['dataset_name'] if 'dataset_name' in self.cv.summary else 'n/a',\n            f'Mean OOF {SCORE_NAME}': mean_oof_score,\n            f'{SCORE_NAME}': score,\n            'R2': R2,\n            'iteration_time': self.cv.summary['iteration_time'] if 'iteration_time' in self.cv.summary else 'n/a',\n        }    \n        return self.cv_scores_        \n\n    def display_cv_results(self):\n        scores = self.cv_scores()\n        print(colored(f'\\n---------- {self.name} {SCORE_NAME}: {scores[SCORE_NAME]} ----------:\\n', 'red'))\n        display(pd.DataFrame([scores,]))\n                \n    def display_cv_plots(self):\n        \n        scores = self.cv_scores()\n        \n        fig, axs = plt.subplots(1, 3, figsize=(20, 10))\n        axs = axs.flatten()\n\n        if 'scores' in self.cv.summary:\n            sns.boxplot(self.cv.summary['scores'], ax=axs[0])\n            axs[0].set_title(f'OOF {SCORE_NAME}')\n        else:\n            axs[0].set_title(f'OOF {SCORE_NAME} N/A')\n\n        df = pd.DataFrame()\n        df['Actual'] = self.cv.oof['true']\n        df['Predicted'] = self.cv.oof['train']\n        sns.scatterplot(df, x='Actual', y='Predicted', ax=axs[1])\n        sns.lineplot(x=[0, np.max(self.cv.oof['train'])], y=[0, np.max(self.cv.oof['train'])], ax=axs[1], color=red)\n        axs[1].set_title('Actual vs Predicted')\n\n        d = PredictionErrorDisplay.from_predictions(np.array(self.cv.oof['true']), np.array(self.cv.oof['train']), ax=axs[2])\n        axs[2].set_title('Prediction error')\n   \n        plt.tight_layout()\n        plt.show()     \n\n    def submit(self):\n        sub = sample_sub.copy()\n        sub[TARGET] = self.cv.oof['test']\n        score = self.cv_scores()[SCORE_NAME]\n        sub.to_csv(f\"{self.name}_{self.cv.summary['name']}_{score:.5f}.csv\", index=False)\n        display(sub.head(30))","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:16:14.67113Z","iopub.execute_input":"2024-12-06T23:16:14.671589Z","iopub.status.idle":"2024-12-06T23:16:14.708828Z","shell.execute_reply.started":"2024-12-06T23:16:14.671555Z","shell.execute_reply":"2024-12-06T23:16:14.707522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass Classifier(Estimator):\n\n    def fit_predict_proba(self, X, y, X_val, y_val):\n        self.fit(X, y, X_val, y_val)\n        return self.predict_proba(X_val)\n        \n    def predict(self, X):\n        assert self.model_ is not None, 'Model error, must be fitted before predict'\n        return np.rint(self.predict_proba(X)).astype(int)\n\n    def predict_proba(self, X):\n        assert self.model_ is not None, 'Model error, must be fitted before predict'\n        return self.model_.predict_proba(X)[:, -1]\n        \n    def crossvalidate(self,\n            train_: pd.DataFrame, \n            test_:  pd.DataFrame,\n            name : str = None,            \n            description : str = '',\n            n_splits=5, n_repeats=1, random_state=42, verbose=2, dataset_name='unknown', use_tqdm=True, clear=True,\n        ) -> CrossValidation :\n\n        if name is None :\n            name = self.name\n        \n        # debug\n        train, test = train_.copy(), test_.copy()\n        train_n_rows = train.shape[0]\n        if verbose > 0 :\n            print(f'\\n---------- cv: train: {train.shape}, test: {test.shape} -----------\\n')\n\n        features = test.columns.to_list()\n\n        folds = RepeatedStratifiedKFold(n_splits=n_splits, n_repeats=n_repeats, random_state=random_state)\n\n        test_proba_mean = np.zeros(len(test))\n        oof_train_proba = np.zeros(len(train))\n        oof_true        = np.zeros(len(train))\n                 \n        oof_scores = []\n\n        assert train_n_rows == len(oof_train_proba) and train_n_rows == len(oof_true), f\"train ({train_n_lows}), oof_test_proba ({len(oof_train_proba)}), oof_true ({len(oof_true)}) doesn't match\"\n        \n        if use_tqdm and verbose > 0:\n            data = tqdm(enumerate(folds.split(train[features], train[TARGET])), desc='Fold', total=n_splits*n_repeats, file=sys.stdout, colour='GREEN')\n        else:\n            data = enumerate(folds.split(train[features], train[TARGET]))\n\n        start = datetime.datetime.now()\n\n        for i, (train_idx,val_idx) in data:\n\n            train_labels =  train.loc[train_idx, TARGET]\n            val_labels =  train.loc[val_idx, TARGET]\n            train_features = train.loc[train_idx, features]\n            val_features = train.loc[val_idx, features]\n    \n            val_proba =  self.fit_predict_proba(train_features, train_labels, val_features, val_labels)\n            test_proba = self.predict_proba(test)\n\n            test_proba_mean += test_proba    \n\n            score = SCORE(val_labels, val_proba)\n            oof_scores.append(score)\n            \n            oof_train_proba[val_idx] = val_proba\n            oof_true[val_idx] = val_labels\n            assert train_n_rows == len(oof_train_proba) and train_n_rows == len(oof_true), f\"train ({train_n_lows}), oof_test_proba ({len(oof_train_proba)}), oof_true ({len(oof_true)}) doesn't match\"\n\n            if clear:\n                clear_output(wait=True)\n \n            if verbose > 0:\n                print('\\nfold', i, SCORE_NAME, score, '\\n')\n\n        if clear:\n            clear_output(wait=True)\n\n        iteration_time = (datetime.datetime.now() - start).total_seconds() / n_splits * n_repeats\n\n        test_proba_mean /= n_splits * n_repeats\n\n        assert train_n_rows == len(oof_train_proba) and train_n_rows == len(oof_true), f\"train ({train_n_lows}), oof_test_proba ({len(oof_train_proba)}), oof_true ({len(oof_true)}) doesn't match\"\n        \n        self.cv = CrossValidation(\n            name                = name,\n            estimator_class     = self.name,\n            dataset_name        = dataset_name,\n            params              = self.params_,\n            n_splits            = n_splits, \n            n_repeats           = n_repeats, \n            \n            oof_train_proba     = oof_train_proba, \n            oof_true            = oof_true, \n            oof_test_proba      = test_proba_mean, \n            \n            oof_scores          = oof_scores,\n            description         = description,\n            iteration_time      = iteration_time,            \n        )\n\n        if verbose > 0:\n            self.display_cv_results()\n        if verbose > 1:\n            self.display_cv_plots()\n        return self.cv\n\n    def decision(self, X) -> np.ndarray :\n        return np.rint(X).astype(int)\n    \n    def cv_scores(self):       \n        oof_train_proba = self.cv.oof['train_proba']\n        oof_true = self.cv.oof['true']\n\n        if 'scores' in self.cv.summary :\n            mean_oof_score = np.mean(self.cv.summary['scores'])\n        else:\n            mean_oof_score = None\n            \n        prediction = self.decision(oof_train_proba)\n        score = SCORE(oof_true, oof_train_proba)\n\n        model_precision, model_recall, model_f1, _ = precision_recall_fscore_support(oof_true, prediction, average=\"weighted\")\n        model_precision, model_recall, model_f1 = round(model_precision, 4), round(model_recall, 4), round(model_f1, 4)\n        model_matthews_corrcoef = round(matthews_corrcoef(oof_true, prediction), 4)\n\n        self.cv_scores_ = {\n            'Model': self.name,\n            'Dataset': self.cv.summary['dataset_name'] if 'dataset_name' in self.cv.summary else 'n/a',\n            f'Mean OOF {SCORE_NAME}': mean_oof_score,\n            f'{SCORE_NAME}': score,\n            'Accuracy': accuracy_score(oof_true, prediction),\n            'Precision Score': model_precision,\n            'Recall Score': model_recall,\n            'f1 Score': model_f1,\n            'Matthews Corr Coef': model_matthews_corrcoef,\n            'iteration_time': self.cv.summary['iteration_time'] if 'iteration_time' in self.cv.summary else 'n/a',\n        }    \n        return self.cv_scores_        \n\n    def display_cv_results(self):\n        scores = self.cv_scores()\n        print(colored(f'\\n---------- {self.name} {SCORE_NAME}: {scores[SCORE_NAME]} ----------:\\n', 'red'))\n        display(pd.DataFrame([scores,]))\n                \n    def display_cv_plots(self):\n        scores = self.cv_scores()\n        fig, axs = plt.subplots(2, 3, figsize=(20, 10))\n        axs = axs.flatten()\n\n        if 'scores' in self.cv.summary:\n            sns.boxplot(self.cv.summary['scores'], ax=axs[0])\n            axs[0].set_title(f'OOF {SCORE_NAME}')\n        else:\n            axs[0].set_title(f'OOF {SCORE_NAME} N/A')\n\n        prediction = self.decision(self.cv.oof['train_proba'])\n        confusion = confusion_matrix(self.cv.oof['true'], prediction)\n        labels = ['class 0', 'Class 1']\n        sns.heatmap(confusion, annot=True, annot_kws={\"fontsize\":24}, fmt=\",d\", xticklabels=labels, yticklabels=labels, cmap='plasma', cbar=False, ax=axs[1])\n        axs[1].set_title(f'Prediction')\n        axs[1].set_ylabel(\"Actual Class\")\n        axs[1].set_xlabel(\"Predicted Class\")    \n\n        RocCurveDisplay.from_predictions(self.cv.oof['true'], self.cv.oof['train_proba'], ax=axs[2])\n        axs[2].set_title('ROC')\n        CalibrationDisplay.from_predictions(self.cv.oof['true'], np.array(self.cv.oof['train_proba']).clip(0, 1), n_bins=30, strategy='quantile', ax=axs[3])\n        axs[3].set_title('Calibration')\n\n        PrecisionRecallDisplay.from_predictions(self.cv.oof['true'], np.array(self.cv.oof['train_proba']).clip(0, 1), ax=axs[4])\n        axs[4].set_title('Precision-Recall')\n        \n        fpr, fnr, thresholds = det_curve(self.cv.oof['true'], self.cv.oof['train_proba'])\n        ax = axs[5]\n        sns.lineplot(x=thresholds, y=fpr, label=f'FPR', ax=ax, color='navy')\n        sns.lineplot(x=thresholds, y=fnr, label=f'FNR', ax=ax, color='red')\n\n        # fr = fpr + fnr\n        # thr = thresholds[fr.argmin()]       \n        \n        # sns.lineplot(x=thresholds, y=fnr+fpr, label=f'FR', ax=ax, color='green')\n        # sns.lineplot(x=[thr, thr], y=[0, fr.min()], ax=ax, color='green')\n        \n        ax.set_title('FPR-FNR curves')\n        ax.set_xlabel('Threshold')\n        ax.set_ylabel('Error Rate')\n\n        plt.tight_layout()\n        plt.show()     \n\n    def submit(self):\n        sub = sample_sub.copy()\n        sub[TARGET] = self.decision(self.cv.oof['test_proba'])\n        score = self.cv_scores()[SCORE_NAME]\n        sub.to_csv(f\"{self.name}_{score}.csv\", index=False)\n        display(sub.head(30))\n        ","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:16:14.710564Z","iopub.execute_input":"2024-12-06T23:16:14.710943Z","iopub.status.idle":"2024-12-06T23:16:14.746245Z","shell.execute_reply.started":"2024-12-06T23:16:14.710909Z","shell.execute_reply":"2024-12-06T23:16:14.745043Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"        \nclass Multiclass(Classifier) :\n\n    def fit_predict_proba(self, X, y, X_val, y_val):\n        self.fit(X, y, X_val, y_val)\n        return self.predict_proba(X_val)\n        \n    def predict(self, X):\n        return np.rint(self.predict_proba(X)).astype(int)\n\n    def predict_proba(self, X):\n        assert self.model_ is not None, 'Model error, must be fitted before predict'\n        return self.model_.predict_proba(X)\n        \n    def crossvalidate(self,\n            train_: pd.DataFrame, \n            test_:  pd.DataFrame,\n            name : str = None,            \n            description : str = '',\n            n_splits=5, n_repeats=1, random_state=42, verbose=2, dataset_name='unknown', use_tqdm=True, clear=True,\n        ) -> CrossValidation :\n\n        if name is None :\n            name = self.name\n        \n        # debug\n        train, test = train_.copy(), test_.copy()\n        train_n_rows = train.shape[0]\n        if verbose > 0 :\n            print(f'\\n---------- cv: train: {train.shape}, test: {test.shape} -----------\\n')\n\n        self.classes = train[TARGET].unique()\n        self.labels = [f'class {i}' for i in self.classes]\n\n        self.target_encoder = LabelEncoder()\n        train[TARGET] = self.target_encoder.fit_transform(train[TARGET])\n\n        features = test.columns.to_list()\n\n        folds = RepeatedStratifiedKFold(n_splits=n_splits, n_repeats=n_repeats, random_state=random_state)\n\n        oof_test = []\n        oof_train = []\n        oof_true = []\n        oof_scores = []\n        \n        if use_tqdm and verbose > 0:\n            data = tqdm(enumerate(folds.split(train[features], train[TARGET])), desc='Fold', total=n_splits*n_repeats, file=sys.stdout, colour='GREEN')\n        else:\n            data = enumerate(folds.split(train[features], train[TARGET]))\n\n        start = datetime.datetime.now()\n\n        for i, (train_idx,val_idx) in data:\n\n            train_labels =  train.loc[train_idx, TARGET]\n            val_labels =  train.loc[val_idx, TARGET]\n            train_features = train.loc[train_idx, features]\n            val_features = train.loc[val_idx, features]\n    \n            val_proba =  self.fit_predict_proba(train_features, train_labels, val_features, val_labels)\n            test_proba = self.predict_proba(test)\n\n            oof_test.append(test_proba)\n\n            val_prediction = np.rint(val_proba).astype(int)\n            score = SCORE(val_labels, self.prediction(val_proba))\n            oof_scores.append(score)\n\n            oof_train.extend(val_proba)\n            oof_true.extend(val_labels)\n\n            if clear:\n                clear_output(wait=True)\n \n            if verbose > 0:\n                print('\\nfold', i, SCORE_NAME, score, '\\n')\n\n        if clear:\n            clear_output(wait=True)\n\n        iteration_time = (datetime.datetime.now() - start).total_seconds() / n_splits * n_repeats\n\n        oof_test = np.mean(oof_test, axis=0)\n        \n        self.cv = CrossValidation(\n            name                = name,\n            estimator_class     = self.name,\n            dataset_name        = dataset_name,\n            params              = self.params_,\n            n_splits            = n_splits, \n            n_repeats           = n_repeats, \n            oof_train           = oof_train, \n            oof_true            = oof_true, \n            oof_test            = oof_test, \n            oof_scores          = oof_scores,\n            description         = description,\n            iteration_time      = iteration_time,            \n        )\n\n        if verbose > 0:\n            self.display_cv_results()\n        if verbose > 1:\n            self.display_cv_plots()\n        return self.cv\n\n    def prediction(self, X : np.ndarray) -> np.ndarray :\n        return np.argmax(X, axis=1)\n    \n    def cv_scores(self):       \n        oof_train_proba = self.cv.oof['train']\n        oof_true = self.cv.oof['true']\n\n        if 'scores' in self.cv.summary :\n            mean_oof_score = np.mean(self.cv.summary['scores'])\n        else:\n            mean_oof_score = None\n            \n        prediction = self.prediction(oof_train_proba).astype(int)\n        score = SCORE(oof_true, self.prediction(oof_train_proba))\n\n#         model_accuracy = round(accuracy_score(oof_true, prediction), 4)\n        model_precision, model_recall, model_f1, _ = precision_recall_fscore_support(oof_true, prediction, average=\"weighted\")\n        model_precision, model_recall, model_f1 = round(model_precision, 4), round(model_recall, 4), round(model_f1, 4)\n        model_matthews_corrcoef = round(matthews_corrcoef(oof_true, prediction), 4)\n\n        self.cv_scores_ = {\n            'Model': self.name,\n            'Dataset': self.cv.summary['dataset_name'] if 'dataset_name' in self.cv.summary else 'n/a',\n            f'Mean OOF {SCORE_NAME}': mean_oof_score,\n            f'{SCORE_NAME}': score,\n#             'Accuracy Score': model_accuracy,\n            'Precision Score': model_precision,\n            'Recall Score': model_recall,\n            'f1 Score': model_f1,\n            'Matthews Corr Coef': model_matthews_corrcoef,\n            'iteration_time': self.cv.summary['iteration_time'] if 'iteration_time' in self.cv.summary else 'n/a',\n        }    \n        return self.cv_scores_             \n               \n    def display_cv_plots(self):\n        scores = self.cv_scores()\n        fig, axs = plt.subplots(1, 2, figsize=(20, 10))\n        axs = axs.flatten()\n\n        if 'scores' in self.cv.summary:\n            sns.boxplot(self.cv.summary['scores'], ax=axs[0])\n            axs[0].set_title(f'OOF {SCORE_NAME}')\n        else:\n            axs[0].set_title(f'OOF {SCORE_NAME} N/A')\n\n        prediction = self.prediction(self.cv.oof['train'])\n        \n        confusion = confusion_matrix(self.cv.oof['true'], prediction)\n        \n        sns.heatmap(confusion, annot=True, annot_kws={\"fontsize\":24}, fmt=\",d\", xticklabels=self.labels, yticklabels=self.labels, cmap='plasma', cbar=False, ax=axs[1])\n        axs[1].set_title(f'Prediction')\n        axs[1].set_ylabel(\"Actual Class\")\n        axs[1].set_xlabel(\"Predicted Class\")    \n\n        plt.tight_layout()\n        plt.show()     \n\n    def submit(self):\n        sub = sample_sub.copy()\n        sub[TARGET] = self.target_encoder.inverse_transform(self.prediction(self.cv.oof['test']))\n        score = self.cv_scores()[SCORE_NAME]\n        sub.to_csv(f\"{self.name}_{score:.5f}.csv\", index=False)\n        display(sub.head(3))\n   ","metadata":{"_kg_hide-input":true,"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-12-06T23:16:14.74789Z","iopub.execute_input":"2024-12-06T23:16:14.74833Z","iopub.status.idle":"2024-12-06T23:16:14.777404Z","shell.execute_reply.started":"2024-12-06T23:16:14.748282Z","shell.execute_reply":"2024-12-06T23:16:14.776257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" \nclass RegressorWrapper(Regressor):\n    \n    def __init__(self, model, name='model', params={}, verbose=1):\n        super().__init__(name=name, params=params, verbose=verbose)\n        self.model_ = model\n        \n    def fit(self, X, y, X_val, y_val):\n        self.model_.fit(X, y)\n\n\nclass RegressorToClassifierWrapper(Classifier):\n    \n    def __init__(self, model, name='model', params={}, verbose=1) :\n        super().__init__(name=name, params=params, verbose=verbose)\n        self.model_ = model\n        \n    def fit(self, X, y, X_val, y_val):\n        self.model_.fit(X, y)\n        \n    def predict_proba(self, X):\n        return self.model_.predict(X)\n         \n    def predict(self, X):\n        assert self.model_ is not None, 'Model error, must be fitted before predict'\n        return DECISION(self.model_.predict(X))\n   \n    \nclass ClassifierWrapper(Classifier):\n    \n    def __init__(self, model, name='model', params={}, verbose=1):\n        super().__init__(name=name, params=params, verbose=verbose)\n        self.model_ = model\n        \n    def fit(self, X, y, X_val, y_val):\n        self.model_.fit(X, y)    \n    \nclass MulticlassWrapper(Multiclass):\n    \n    def __init__(self, model, name='model', params={}, verbose=1):\n        super().__init__(name=name, params=params, verbose=verbose)\n        self.model_ = model\n        \n    def fit(self, X, y, X_val, y_val):\n        self.model_.fit(X, y)","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:16:14.782338Z","iopub.execute_input":"2024-12-06T23:16:14.782752Z","iopub.status.idle":"2024-12-06T23:16:14.801613Z","shell.execute_reply.started":"2024-12-06T23:16:14.782712Z","shell.execute_reply":"2024-12-06T23:16:14.800526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass EnsembleRegressor(Regressor):\n    \n    def __init__(self, estimators, name='ENS', params={}, verbose=1, options = {}):\n        super().__init__(name=name, params=params, verbose=verbose)\n        \n        self.estimators_ = estimators\n        self.options_ = options\n        \n    def fit_predict(self, X, y, X_val, y_val):\n        \n        print('\\n')\n\n        self.models_ = []\n\n        val_preds = []\n\n        for estimator in self.estimators_:\n            try:\n                m = clone(estimator)\n            except:\n                m = estimator\n            val_p = m.fit_predict(X, y, X_val, y_val)\n            if self.verbose_ > 0:\n                print(colored(f'\\n{m.__class__.__name__}: {SCORE(y_val, val_p)}', 'blue'))\n            self.models_.append(m)\n            val_preds.append(val_p)\n        \n        if self.verbose_ > 0:\n            print('\\nPREDS:')\n            for pred in val_preds:\n                print(pred)\n\n            preds_mean = np.mean(val_preds, axis=0)\n            print('MEAN:\\n', preds_mean)\n\n            print(colored(f'{SCORE_NAME} OF MEAN: {SCORE(y_val, preds_mean)}', 'red'))\n\n        self.averager_ = Averager()\n        VAL_PREDS = self.averager_.fit_predict(val_preds, y_val)\n        if self.verbose_ > 0:\n            print('\\nWEIGHTS:\\n', self.averager_.weights())\n            print('AVGW (weighted average):\\n', VAL_PREDS)\n\n            print(colored(f'{SCORE_NAME} OF AVGW: {SCORE(y_val, VAL_PREDS)}\\n\\n', 'red'))\n        \n        if 'optimize' in self.options_:\n            if not self.options_['optimize']:\n                return preds_mean\n        return VAL_PREDS\n            \n    def predict(self, X):\n        assert len(self.models_) > 0, 'Model error, must be fitted before predict'\n        if len(self.models_) == 1:\n            return self.models_[0].predict(X)\n\n        preds = [model.predict(X) for model in self.models_]\n        \n        if 'optimize' in self.options_:\n            if not self.options_['optimize']:\n                return np.mean(preds, axis=0)\n        return self.averager_.predict(preds)       \n    \n# EXAMPLE\nestimator = EnsembleRegressor(\n    [\n        RegressorWrapper(LGBMRegressor(n_estimators=100, random_state=42, verbose=-1)) ,\n        RegressorWrapper(XGBRegressor(n_estimators=100, random_state=42, enable_categorical=True)) ,\n    ], \n    name='ENS',\n)","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:16:14.803123Z","iopub.execute_input":"2024-12-06T23:16:14.803584Z","iopub.status.idle":"2024-12-06T23:16:14.818765Z","shell.execute_reply.started":"2024-12-06T23:16:14.80353Z","shell.execute_reply":"2024-12-06T23:16:14.817349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass EnsembleClassifier(Classifier):\n    \n    def __init__(self, classifiers, name='ENS', params={}, verbose=1, options = {}):\n        super().__init__(name=name, params=params, verbose=verbose)\n        \n        self.classifiers_ = classifiers\n        self.options_ = options\n        \n    def fit_predict_proba(self, X, y, X_val, y_val):\n        \n        print('\\n')\n\n        self.models_ = []\n\n        val_probas = []\n\n        for classifier in self.classifiers_:\n            try:\n                m = clone(classifier)\n            except:\n                m = classifier\n            val_p = m.fit_predict_proba(X, y, X_val, y_val)\n            if self.verbose_ > 0:\n                print(colored(f'\\n{m.__class__.__name__}: {-LOSS(y_val, val_p)}', 'blue'))\n            self.models_.append(m)\n            val_probas.append(val_p)\n        \n        if self.verbose_ > 0:\n            print('\\nPROBAS:')\n            for proba in val_probas:\n                print(proba)\n\n            probas_mean = np.mean(val_probas, axis=0)\n            print('MEAN:\\n', probas_mean)\n\n            print(colored(f'-LOSS OF MEAN: {-LOSS(y_val, probas_mean)}', 'red'))\n\n        self.averager_ = Averager()\n        VAL_PROBAS = self.averager_.fit_predict(val_probas, y_val)\n        if self.verbose_ > 0:\n            print('\\nWEIGHTS:\\n', self.averager_.weights())\n            print('AVGW (weighted average):\\n', VAL_PROBAS)\n\n            print(colored(f'-LOSS OF AVGW: {-LOSS(y_val, VAL_PROBAS)}\\n\\n', 'red'))\n        \n        if 'optimize' in self.options_:\n            if not self.options_['optimize']:\n                return probas_mean\n        return VAL_PROBAS\n\n    def predict(self, X) :\n        return self.decision(self.predict_proba(X))        \n            \n    def predict_proba(self, X):\n        assert len(self.models_) > 0, 'Model error, must be fitted before predict'\n        if len(self.models_) == 1:\n            return self.models_[0].predict_proba(X)\n\n        probas = [model.predict_proba(X) for model in self.models_]\n        \n        if 'optimize' in self.options_:\n            if not self.options_['optimize']:\n                return np.mean(probas, axis=0)\n        return self.averager_.predict(probas)       \n    \n# EXAMPLE\nestimator = EnsembleClassifier(\n    [\n        ClassifierWrapper(LGBMClassifier(n_estimators=100, random_state=42, verbose=-1, objective='binary', metric='auc')) ,\n        ClassifierWrapper(XGBClassifier(n_estimators=100, random_state=42, enable_categorical=True, objective='binary:logistic')) ,\n    ], \n    name='ENS',\n)","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:16:14.820508Z","iopub.execute_input":"2024-12-06T23:16:14.821225Z","iopub.status.idle":"2024-12-06T23:16:14.840431Z","shell.execute_reply.started":"2024-12-06T23:16:14.82115Z","shell.execute_reply":"2024-12-06T23:16:14.83909Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class StackRegressor(Regressor):\n    \n    def __init__(self, \n            estimators : list,\n            datasets_names : list,\n            final_estimator: Estimator,\n            name : str = 'STK', \n            params : dict = {}, \n            verbose : int = 2,\n        ):\n        \n        super().__init__(name = name, params = params, verbose = verbose)\n        \n        self.estimators_ = estimators \n        self.datasets_names_ = datasets_names\n        self.final_estimator_ = final_estimator\n\n        count = len(self.estimators_)\n        if type(self.datasets_names_) is str :\n            self.datasets_names_ = [self.datasets_names_] * count\n        \n    def crossvalidate(self,\n            name : str = None,            \n            description : str = '',\n            n_splits=5, n_repeats=1, random_state=42, verbose=2, use_tqdm=True, clear=True,\n        ) -> CrossValidation :\n\n        if name is None :\n            name = self.name\n        y = DB.datasets.train(self.datasets_names_[0])[TARGET]\n        trn = pd.DataFrame()\n        tst = pd.DataFrame()\n        trn[TARGET] = y\n        for i, estimator in enumerate(self.estimators_) :\n            yi = DB.datasets.train(self.datasets_names_[i])[TARGET]\n            assert np.array_equal(y, yi, equal_nan=True), 'datasets are not equal'\n            cv = estimator.crossvalidate(\n                    DB.datasets.train(self.datasets_names_[i]),\n                    DB.datasets.test(self.datasets_names_[i]),\n                    dataset_name = self.datasets_names_[i], \n                    n_splits = n_splits,\n                    n_repeats = n_repeats,\n                    clear = False,\n                    verbose = 1,\n                    name = '',\n            )\n            trn[f'est{i}'] = cv.oof['train']\n            tst[f'est{i}'] = cv.oof['test']\n\n        display(trn)\n\n        display(tst)\n            \n        self.cv = self.final_estimator_.crossvalidate(\n                trn, tst,\n                dataset_name = str(self.datasets_names_), \n                n_splits = 10,\n                n_repeats = 1,\n                clear = clear,\n                use_tqdm = use_tqdm,\n                verbose = verbose,\n                name = name,\n        )\n        return   self.cv\n\n# EXAMPLE\nestimator = StackRegressor(\n    estimators = [\n        RegressorWrapper(LGBMRegressor(n_estimators=100, random_state=42, verbose=-1), name = 'LGBM'),\n        RegressorWrapper(XGBRegressor(n_estimators=100, random_state=42, enable_categorical=True), name = 'XGB'),\n    ],\n    datasets_names = 'lxe',\n    final_estimator = RegressorWrapper(LinearRegression(), name='LR'),\n    name = 'STK_LR'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:16:14.842359Z","iopub.execute_input":"2024-12-06T23:16:14.843113Z","iopub.status.idle":"2024-12-06T23:16:14.863711Z","shell.execute_reply.started":"2024-12-06T23:16:14.84306Z","shell.execute_reply":"2024-12-06T23:16:14.862517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass StackClassifier(Classifier):\n    \n    def __init__(self, \n            estimators : list,\n            datasets_names : list,\n            final_estimator: Estimator,\n            name : str = 'STK', \n            params : dict = {}, \n            verbose : int = 2,\n        ):\n        \n        super().__init__(name = name, params = params, verbose = verbose)\n        \n        self.estimators_ = estimators \n        self.datasets_names_ = datasets_names\n        self.final_estimator_ = final_estimator\n\n        count = len(self.estimators_)\n        if type(self.datasets_names_) is str :\n            self.datasets_names_ = [self.datasets_names_] * count\n        \n    def crossvalidate(self,\n            name : str = None,            \n            description : str = '',\n            n_splits=5, n_repeats=1, random_state=42, verbose=2, use_tqdm=True, clear=True,\n        ) -> CrossValidation :\n\n        if name is None :\n            name = self.name\n        y = DB.datasets.train(self.datasets_names_[0])[TARGET]\n        trn = pd.DataFrame()\n        tst = pd.DataFrame()\n        trn[TARGET] = y\n        for i, estimator in enumerate(self.estimators_) :\n            yi = DB.datasets.train(self.datasets_names_[i])[TARGET]\n            assert np.array_equal(y, yi, equal_nan=True), 'datasets are not equal'\n            cv = estimator.crossvalidate(\n                    DB.datasets.train(self.datasets_names_[i]),\n                    DB.datasets.test(self.datasets_names_[i]),\n                    dataset_name = self.datasets_names_[i], \n                    n_splits = n_splits,\n                    n_repeats = n_repeats,\n                    clear = False,\n                    verbose = 1,\n                    name = '',\n            )\n            trn[f'est{i}'] = cv.oof['train']\n            tst[f'est{i}'] = cv.oof['test']\n\n        display(trn)\n\n        display(tst)\n            \n        self.cv = self.final_estimator_.crossvalidate(\n                trn, tst,\n                dataset_name = str(self.datasets_names_), \n                n_splits = 10,\n                n_repeats = 1,\n                clear = clear,\n                use_tqdm = use_tqdm,\n                verbose = verbose,\n                name = name,\n        )\n        return   self.cv\n\n# EXAMPLE\nestimator = StackClassifier(\n    estimators = [\n        ClassifierWrapper(LGBMClassifier(n_estimators=100, random_state=42, verbose=-1, objective='binary', metric='auc'), name = 'LGBM'),\n        ClassifierWrapper(XGBClassifier(n_estimators=100, random_state=42, enable_categorical=True, objective='binary:logistic'), name = 'XGB'),\n    ],\n    datasets_names = 'lxe',\n    final_estimator = ClassifierWrapper(LogisticRegression(penalty='l2',  C=10, max_iter=5000, tol=1e-6), name='LR'),\n    name = 'STK_LR'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:16:14.86568Z","iopub.execute_input":"2024-12-06T23:16:14.866115Z","iopub.status.idle":"2024-12-06T23:16:14.88666Z","shell.execute_reply.started":"2024-12-06T23:16:14.866071Z","shell.execute_reply":"2024-12-06T23:16:14.88526Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# &#128295; Execution settings","metadata":{}},{"cell_type":"code","source":"EVALUATE_DATASETS = True\nEVALUATE_MODELS   = True\nEDA               = True\nSUBMIT            = True\nOPTUNA            = False\nDEBUG             = False\n\nDB                = Database('DATABASE (1)')\nLOAD_DB           = True\nSHOW_CVs          = True\n\nif LOAD_DB :\n    DB.load()\nDB","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:16:14.888615Z","iopub.execute_input":"2024-12-06T23:16:14.889083Z","iopub.status.idle":"2024-12-06T23:18:06.853275Z","shell.execute_reply.started":"2024-12-06T23:16:14.889038Z","shell.execute_reply":"2024-12-06T23:18:06.852054Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# &#128204; *Models*","metadata":{}},{"cell_type":"code","source":"ESTIMATORS = {} # for cv load / rerun","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:06.854492Z","iopub.execute_input":"2024-12-06T23:18:06.85479Z","iopub.status.idle":"2024-12-06T23:18:06.859688Z","shell.execute_reply.started":"2024-12-06T23:18:06.854762Z","shell.execute_reply":"2024-12-06T23:18:06.858571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if AUTOGLUON :\n    \n    from autogluon.tabular import TabularPredictor\n    \n    class AGClassifier(Classifier):\n    \n        def __init__(self, name : str = 'AGC', params : dict = {}, verbose : int = 1, time_limit : int = 300, eval_metric : str = 'roc_auc') :\n            super().__init__(name=name, params=params, verbose=verbose)\n            \n            self.time_limit_ = time_limit\n            self.eval_metric_ = eval_metric\n            \n        def fit(self, X, y, X_val, y_val):\n            X[TARGET]  = y\n            \n            self.model_ = TabularPredictor(\n                label = TARGET,\n                eval_metric = self.eval_metric_,\n                problem_type = 'binary',\n            ).fit(\n                X,\n                presets = 'best_quality',\n                time_limit = self.time_limit_,\n                verbosity = self.verbose_,\n                ag_args_fit = {'num_cpus': 4},\n            )\n            \n        def predict(self, X):\n            pred = self.model_.predict(X)\n            print('pred:', pred)\n            return pred    \n            \n        def predict_proba(self, X):\n            proba = self.model_.predict_proba(X)\n            print('proba:', proba)\n            return proba.loc[:, 1]    \n        \n    \n    class AGRegressor(Regressor):\n    \n        def __init__(self, name : str = 'AGR', params : dict = {}, verbose : int = 1, time_limit : int = 300, eval_metric : str = 'root_mean_squared_error') :\n            super().__init__(name=name, params=params, verbose=verbose)\n            \n            self.time_limit_ = time_limit\n            self.eval_metric_ = eval_metric\n            \n        def fit(self, X, y, X_val, y_val):\n            X[TARGET]  = y\n            \n            self.model_ = TabularPredictor(\n                label = TARGET,\n                eval_metric = self.eval_metric_,\n                problem_type = 'regression',\n            ).fit(\n                X,\n                presets = 'best_quality',\n                time_limit = self.time_limit_,\n                verbosity = self.verbose_,\n                ag_args_fit = {'num_cpus': 4},\n            )\n            \n    ESTIMATORS['AGC'] = AGClassifier\n    ESTIMATORS['AGR'] = AGRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:06.861265Z","iopub.execute_input":"2024-12-06T23:18:06.861707Z","iopub.status.idle":"2024-12-06T23:18:06.878536Z","shell.execute_reply.started":"2024-12-06T23:18:06.861656Z","shell.execute_reply":"2024-12-06T23:18:06.877419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass LGBRegressor(Regressor):\n\n    def __init__(self, name='LGBR', params={}, verbose=1) :\n        super().__init__(name=name, params=params, verbose=verbose)\n        \n    def fit(self, X, y, X_val, y_val):\n        self.model_ = LGBMRegressor(**self.params_)\n        self.model_.fit(\n            X, y, \n            eval_set=(X_val, y_val),\n            callbacks = [log_evaluation(period=100, show_stdv=False), early_stopping(stopping_rounds=200, verbose=self.verbose_)],            \n        )      \n    \n\nclass LGBClassifier(Classifier):\n\n    def __init__(self, name='LGBC', params={}, verbose=1) :\n        super().__init__(name=name, params=params, verbose=verbose)\n        \n    def fit(self, X, y, X_val, y_val):\n        self.model_ = LGBMClassifier(**self.params_)\n        self.model_.fit(\n            X, y, \n            eval_set=(X_val, y_val),\n            callbacks = [log_evaluation(period=100, show_stdv=False), early_stopping(stopping_rounds=200, verbose=self.verbose_)],            \n        )      \n\n    \nESTIMATORS['LGBC'] = LGBClassifier\nESTIMATORS['LGBR'] = LGBRegressor    ","metadata":{"_kg_hide-input":false,"scrolled":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:06.880067Z","iopub.execute_input":"2024-12-06T23:18:06.880466Z","iopub.status.idle":"2024-12-06T23:18:06.898666Z","shell.execute_reply.started":"2024-12-06T23:18:06.88043Z","shell.execute_reply":"2024-12-06T23:18:06.897436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CATRegressor(Regressor):\n\n    def __init__(self, name='CATR', params={}, verbose=1) :\n        super().__init__(name=name, params=params, verbose=verbose)\n\n        self.evals_results_ = []\n        \n    def fit(self, X, y, X_val = None, y_val = None) :\n        if X_val is not None :\n            cat_features = fcn(X)[1] # X.columns.tolist()            \n            train_pool = Pool(data=X, label=y, cat_features=cat_features)\n            valid_pool = Pool(data=X_val, label=y_val, cat_features=cat_features)\n            self.model_ = catboost.CatBoostRegressor(**self.params_)\n            self.model_.fit(train_pool, eval_set=valid_pool, verbose=20)\n        else:\n            cat_features = X.columns.tolist()            \n            train_pool = Pool(data=X, label=y, cat_features=cat_features)\n            params =  self.model_.get_params()\n            params.update({\n                'iterations'     : int(self.model_.tree_count_ * 1.2),\n            })\n    \n            new_params = {}\n            for k, v in params.items() :\n                if not k in ['use_best_model', 'od_type', 'od_wait', 'early_stopping_rounds'] :\n                    new_params[k] = v\n                    \n            print('\\n--- FIT ---', new_params, '---\\n')\n                    \n            self.model_ = catboost.CatBoostRegressor(**new_params)\n            self.model_.fit(train_pool, verbose=200)   \n\n        self.evals_results_.append(self.model_.get_evals_result())\n\n    def fit_predict(self, X, y, X_val, y_val):\n        self.fit(X, y, X_val, y_val)\n        return self.predict(X_val)\n        \n    def predict(self, X):\n        best_iter = self.model_.best_iteration_\n        return self.model_.predict(X, ntree_end=best_iter)\n\n    def plot_metrics_(self) :\n        if self.evals_results_  == [] :\n            print('no metrics to plot')\n            return\n    \n        data, metrics_names = [], set()\n        for fold, results in enumerate(self.evals_results_) :\n            for curve, metrics in results.items() :\n                for metric, values in metrics.items() :\n                    metrics_names.add(metric)\n                    for iter, value in enumerate(values) :\n                        item = {\n                            'fold'      : fold,\n                            'iteration' : iter,\n                            'curve'     : curve,\n                             metric     : value,\n                        }   \n                        data.append(item)\n    \n        df = pd.DataFrame(data)\n        nrows = len(metrics_names)\n        fig, ax = plt.subplots(nrows, 1, figsize = (16, 6 * nrows))\n        if nrows > 1 :\n            ax = ax.flatten()\n        else :\n            ax = [ax]\n        for i, metric in enumerate(metrics_names) :\n            sns.lineplot(df, x = 'iteration', y = metric, hue = 'curve', style = 'curve', markers = False, ax = ax[i])\n            ax[i].set_ylabel(metric)\n            # ax[i].set_xticks(range(df['epoch'].max() + 1))\n        plt.tight_layout()\n        plt.show()\n    \n    def display_cv_plots(self) -> None :  \n        super().display_cv_plots()\n        self.plot_metrics_()\n        \nESTIMATORS['CATR'] = CATRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:06.900439Z","iopub.execute_input":"2024-12-06T23:18:06.900889Z","iopub.status.idle":"2024-12-06T23:18:06.917874Z","shell.execute_reply.started":"2024-12-06T23:18:06.900842Z","shell.execute_reply":"2024-12-06T23:18:06.91683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CATClassifier(Classifier):\n\n    def __init__(self, name='CATC', params={}, verbose=1) :\n        super().__init__(name=name, params=params, verbose=verbose)\n\n        self.evals_results_ = []\n        \n    def fit(self, X, y, X_val = None, y_val = None) :\n        if X_val is not None :\n            if 'cat_features' in self.params_ :\n                cat_features = self.params_['cat_features']\n            else :\n                cat_features = X.columns.tolist()            \n            train_pool = Pool(data=X, label=y, cat_features=cat_features)\n            valid_pool = Pool(data=X_val, label=y_val, cat_features=cat_features)\n            self.model_ = catboost.CatBoostClassifier(**self.params_)\n            self.model_.fit(train_pool, eval_set=valid_pool, verbose=200)\n        else:\n            cat_features = X.columns.tolist()            \n            train_pool = Pool(data=X, label=y, cat_features=cat_features)\n            params =  self.model_.get_params()\n            params.update({\n                'iterations'     : int(self.model_.tree_count_ * 1.2),\n            })\n    \n            new_params = {}\n            for k, v in params.items() :\n                if not k in ['use_best_model', 'od_type', 'od_wait', 'early_stopping_rounds'] :\n                    new_params[k] = v\n                    \n            print('\\n--- FIT ---', new_params, '---\\n')\n                    \n            self.model_ = catboost.CatBoostClassifier(**new_params)\n            self.model_.fit(train_pool, verbose=200)   \n\n        self.evals_results_.append(self.model_.get_evals_result())\n\n    def fit_predict_proba(self, X, y, X_val, y_val):\n        self.fit(X, y, X_val, y_val)\n        return self.predict_proba(X_val)\n        \n    def predict_proba(self, X):\n        best_iter = self.model_.best_iteration_\n        return self.model_.predict_proba(X, ntree_end=best_iter)[:,-1]\n\n    def plot_metrics_(self) :\n        if self.evals_results_  == [] :\n            print('no metrics to plot')\n            return\n    \n        data, metrics_names = [], set()\n        for fold, results in enumerate(self.evals_results_) :\n            for curve, metrics in results.items() :\n                for metric, values in metrics.items() :\n                    metrics_names.add(metric)\n                    for iter, value in enumerate(values) :\n                        item = {\n                            'fold'      : fold,\n                            'iteration' : iter,\n                            'curve'     : curve,\n                             metric     : value,\n                        }   \n                        data.append(item)\n    \n        df = pd.DataFrame(data)\n        nrows = len(metrics_names)\n        fig, ax = plt.subplots(nrows, 1, figsize = (16, 6 * nrows))\n        if nrows > 1 :\n            ax = ax.flatten()\n        else :\n            ax = [ax]\n        for i, metric in enumerate(metrics_names) :\n            sns.lineplot(df, x = 'iteration', y = metric, hue = 'curve', style = 'curve', markers = False, ax = ax[i])\n            ax[i].set_ylabel(metric)\n            # ax[i].set_xticks(range(df['epoch'].max() + 1))\n        plt.tight_layout()\n        plt.show()\n    \n    def display_cv_plots(self) -> None :  \n        super().display_cv_plots()\n        self.plot_metrics_()\n\nESTIMATORS['CATC'] = CATClassifier","metadata":{"_kg_hide-input":false,"scrolled":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:06.919068Z","iopub.execute_input":"2024-12-06T23:18:06.919415Z","iopub.status.idle":"2024-12-06T23:18:06.937427Z","shell.execute_reply.started":"2024-12-06T23:18:06.919378Z","shell.execute_reply":"2024-12-06T23:18:06.936325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\n\nclass XGB():\n        \n    def fit(self, X, y, X_val, y_val):\n        dtrain = xgb.DMatrix(X.to_numpy(), label=y.to_numpy())\n        dvalid = xgb.DMatrix(X_val.to_numpy(), label=y_val.to_numpy())\n        eval_set = [(dtrain, 'train'), (dvalid, 'valid')]        \n        self.model_ = xgb.train(\n            params=self.params_,\n            dtrain=dtrain,\n            num_boost_round=5000,\n            maximize=True,\n            evals=eval_set,\n            early_stopping_rounds=300,\n            verbose_eval=200,\n        )\n        \n    def predict(self, X):\n        return np.clip(self.model_.predict(xgb.DMatrix(X.to_numpy())), a_min = 0,a_max = None)\n\n    def predict_proba(self, X):\n        return self.model_.predict(xgb.DMatrix(X.to_numpy()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:06.938712Z","iopub.execute_input":"2024-12-06T23:18:06.93905Z","iopub.status.idle":"2024-12-06T23:18:06.955358Z","shell.execute_reply.started":"2024-12-06T23:18:06.939017Z","shell.execute_reply":"2024-12-06T23:18:06.954225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class xGBClassifier(Classifier, XGB):\n\n    def __init__(self, name='XGBC', params={}, verbose=1) :\n        Classifier.__init__(self, name=name, params=params, verbose=verbose)\n        \n    def fit(self, X, y, X_val, y_val):\n        XGB.fit(self, X, y, X_val, y_val)\n              \n    def predict_proba(self, X):\n        return XGB.predict_proba(self, X)\n    \n    \nclass xGBRegressor(Regressor, XGB):\n\n    def __init__(self, name='XGBR', params={}, verbose=1) :\n        Regressor.__init__(self, name=name, params=params, verbose=verbose)\n        \n    def fit(self, X, y, X_val, y_val):\n        XGB.fit(self, X, y, X_val, y_val)\n              \n    def predict(self, X):\n        return XGB.predict(self, X)    \nESTIMATORS['XGBC'] = xGBClassifier\nESTIMATORS['XGBR'] = xGBRegressor    ","metadata":{"_kg_hide-input":false,"scrolled":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:06.956979Z","iopub.execute_input":"2024-12-06T23:18:06.957381Z","iopub.status.idle":"2024-12-06T23:18:06.972335Z","shell.execute_reply.started":"2024-12-06T23:18:06.957348Z","shell.execute_reply":"2024-12-06T23:18:06.971134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fcn(df, cat_types = ['object', 'category', 'bool', 'string']):\n    return df.columns.tolist(), df.select_dtypes(include=cat_types).columns.tolist(), df.select_dtypes(exclude=cat_types).columns.tolist()   ","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:06.973651Z","iopub.execute_input":"2024-12-06T23:18:06.973951Z","iopub.status.idle":"2024-12-06T23:18:06.98599Z","shell.execute_reply.started":"2024-12-06T23:18:06.973922Z","shell.execute_reply":"2024-12-06T23:18:06.984894Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATASETS_SCORES = {}","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:06.987394Z","iopub.execute_input":"2024-12-06T23:18:06.987789Z","iopub.status.idle":"2024-12-06T23:18:06.998831Z","shell.execute_reply.started":"2024-12-06T23:18:06.987755Z","shell.execute_reply":"2024-12-06T23:18:06.997667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def display_scores(scores, x='CV', y=SCORE_NAME, plot=True):\n    if scores == {}:\n        print('scores list is empty')\n        return\n    print('\\nCOMPARE:')\n    \n    data = []\n    for cv_, cv_scores_ in scores.items() :\n        item = {'CV' : cv_}\n        for k, v in cv_scores_.items():\n            item[k] = v\n        data.append(item)\n        \n    df = pd.DataFrame(data)\n    display(df.style.background_gradient(subset=df.select_dtypes(include=[float, int]).columns.to_list(), cmap='Greens'))\n\n    if not plot:\n        return\n    mn, mx = df[y].min(), df[y].max()\n    fig, ax = plt.subplots(1, 1, figsize=(15, 5))\n    sns.barplot(df, x=y, y=x, ax=ax)\n    ax.set(xlim=(mn*0.999, mx*1.001))\n    ax.bar_label(ax.containers[0])\n    plt.tight_layout()\n    plt.show()","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:07.00002Z","iopub.execute_input":"2024-12-06T23:18:07.000372Z","iopub.status.idle":"2024-12-06T23:18:07.011702Z","shell.execute_reply.started":"2024-12-06T23:18:07.000341Z","shell.execute_reply":"2024-12-06T23:18:07.01051Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def tocat(src, dst):     \n    train, test = DB.datasets.train(src), DB.datasets.test(src)\n    cats = fcn(test)[1]\n    for df in [train, test] :\n        df[cats] = df[cats].astype('string').fillna('--missed--')\n    DB.datasets.put(dst, train, test)\n\ndef allcat(src, dst):     \n    train, test = DB.datasets.train(src), DB.datasets.test(src)\n    cats = fcn(test)[0]\n    for df in [train, test] :\n        df[cats] = df[cats].astype('string').fillna('--missed--')\n    DB.datasets.put(dst, train, test)\n\ndef enccat(src, dst):     \n    train, test = DB.datasets.train(src), DB.datasets.test(src)\n    _, CATS, NUMS = fcn(test)\n    oe = OrdinalEncoder(handle_unknown='use_encoded_value', unknown_value=-1)\n    \n    X_oe = pd.DataFrame(oe.fit_transform(train[CATS]), columns = CATS).fillna(0).astype(int)\n    \n    test_oe = pd.DataFrame(oe.transform(test[CATS]), columns = CATS).fillna(0).astype(int)\n    \n    train_e = pd.concat([X_oe, train[NUMS]], axis= 1)\n    test_e = pd.concat([test_oe, test[NUMS]], axis= 1)\n    \n    train_e[TARGET] = train[TARGET]    \n    DB.datasets.put(dst, train_e, test_e)\n\n\ndef pipe(src, dst, funcs):\n    tmp = src\n    for func in funcs :\n        print(tmp, func.__name__, dst)\n        func(tmp, dst)\n        tmp = dst","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:07.012956Z","iopub.execute_input":"2024-12-06T23:18:07.013368Z","iopub.status.idle":"2024-12-06T23:18:07.030523Z","shell.execute_reply.started":"2024-12-06T23:18:07.013333Z","shell.execute_reply":"2024-12-06T23:18:07.02949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate_dataset(name, clear=True, db=DB, update=False):\n    tmp = 'tmp'\n    if EVALUATE_DATASETS:\n            \n        params = {\n            'random_state'       : 42,\n        }\n        estimator = LGBRegressor(params=params)\n\n        cv_name = f'dataset_eval_{name}'\n\n        cv = None\n        if db is not None and not update:\n            cv = db.cvs.get(cv_name)\n            if cv is not None:\n                if cv.summary['estimator_class'] != estimator.name or cv.summary['params'] != params :\n                    cv = None\n                else :\n                    estimator.upload_cv(cv)\n                    estimator.display_cv_results()\n                    estimator.display_cv_plots()\n        if cv is None :\n            print('encoding categorical features ...', end=' ')\n            enccat(name, tmp)\n            print(name, 'dataset evaluation ...', end=' ')\n            cv = estimator.crossvalidate(\n                DB.datasets.train(tmp), \n                DB.datasets.test(tmp), \n                n_splits = 3, \n                dataset_name=name, \n                clear=clear, \n                verbose=2,\n                name=cv_name,\n            )\n            if db is not None:\n                print('saving ...', end=' ')\n                db.cvs.put(cv)\n                # db.save()\n                print('ready\\n')\n        \n        DATASETS_SCORES[name] = estimator.cv_scores()\n        \n        # display_scores(DATASETS_SCORES, plot=False)","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:07.031675Z","iopub.execute_input":"2024-12-06T23:18:07.031991Z","iopub.status.idle":"2024-12-06T23:18:07.045255Z","shell.execute_reply.started":"2024-12-06T23:18:07.031962Z","shell.execute_reply":"2024-12-06T23:18:07.043858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def summary(df, tab=True, plots=False, info=True, nrows=3):\n    if tab:\n        display(df.head(nrows))\n        display(df.tail(nrows))\n        print(colored(f'dataframe has {df.shape[0]} rows, {df.shape[1]} columns\\n', 'blue'))\n\n    inf = pd.DataFrame(df.dtypes).reset_index().rename(columns={'index':'column', 0:'type'})\n    df_missed = pd.DataFrame(df.isnull().sum()).reset_index().rename(columns={'index':'column', 0:'missed'})\n    df_unique = pd.DataFrame(df.nunique()).reset_index().rename(columns={'index':'column', 0:'unique'})\n    inf['missed'] = df_missed['missed']\n    inf['unique'] = df_unique['unique']\n    inf['duplicate'] = df.duplicated().sum()\n    \n    desc = pd.DataFrame(df.describe(include='all').transpose())\n    if 'min' in desc.columns.tolist():\n        inf['min'] = desc['min'].values\n        inf['max'] = desc['max'].values\n        inf['avg'] = desc['mean'].values\n        inf['std dev'] = desc['std'].values\n    if 'top' in desc.columns.tolist():\n        inf['top value'] = desc['top'].values\n        inf['Freq'] = desc['freq'].values    \n    \n    if info:\n        display(inf.style.background_gradient(subset='missed', cmap='Reds').background_gradient(subset='unique', cmap='Greens'))\n  \n    if plots:\n        print()\n        if df_missed['missed'].sum() > 0:\n            fig, ax = plt.subplots(1, 1, figsize=(24, 5))\n            sns.barplot(df_missed[df_missed['missed'] > 0], x='column', y='missed', ax=ax)\n            ax.set_title('missed values') \n            ax.bar_label(ax.containers[0])\n            plt.tight_layout()\n            plt.show()\n\n        fig, ax = plt.subplots(1, 1, figsize=(24, 5))\n        sns.barplot(df_unique[df_unique['unique'] > 0], x='column', y='unique', ax=ax)\n        ax.set_title('unique values')\n        ax.bar_label(ax.containers[0])\n        plt.tight_layout()\n        plt.show()\n\n!pip install dython\nfrom dython.nominal import associations\nclear_output()\n\ndef df_corr(df, name=''):\n    associations_df = associations(df, nominal_columns='all', plot=False)\n    corr_matrix = associations_df['corr']\n    plt.figure(figsize=(20, 8))\n    plt.gcf().set_facecolor('#FFFDD0') \n    sns.heatmap(corr_matrix, annot=True, fmt='.2f', cmap='coolwarm', linewidths=0.5)\n    plt.title(f'{name} Correlation Matrix including Categorical Features')\n    plt.show()\n","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:07.050639Z","iopub.execute_input":"2024-12-06T23:18:07.05105Z","iopub.status.idle":"2024-12-06T23:18:17.769166Z","shell.execute_reply.started":"2024-12-06T23:18:07.051015Z","shell.execute_reply":"2024-12-06T23:18:17.76782Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# &#128269; LOAD DATASETS","metadata":{}},{"cell_type":"code","source":"train, test, original, sample_sub = load_datasets()\nDB.datasets.put('loaded', train, test)","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:17.770916Z","iopub.execute_input":"2024-12-06T23:18:17.771306Z","iopub.status.idle":"2024-12-06T23:18:32.425165Z","shell.execute_reply.started":"2024-12-06T23:18:17.771268Z","shell.execute_reply":"2024-12-06T23:18:32.42389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DB.datasets.summary('loaded', plots = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:32.426565Z","iopub.execute_input":"2024-12-06T23:18:32.427003Z","iopub.status.idle":"2024-12-06T23:18:45.354554Z","shell.execute_reply.started":"2024-12-06T23:18:32.426966Z","shell.execute_reply":"2024-12-06T23:18:45.353358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if original is not None:\n    summary(original, plots=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:45.356146Z","iopub.execute_input":"2024-12-06T23:18:45.356531Z","iopub.status.idle":"2024-12-06T23:18:47.808235Z","shell.execute_reply.started":"2024-12-06T23:18:45.356497Z","shell.execute_reply":"2024-12-06T23:18:47.80724Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# &#128202; EDA","metadata":{}},{"cell_type":"markdown","source":"<div style=\"display: block; padding: 3px 0; background-color: #ccffcc; color: black; font-size: 12px; font-weight: bold; border-radius: 12px; text-align: center; width: 100%; box-shadow: inset 4px 4px 8px rgba(255, 255, 255, 0.5), inset -4px -4px 8px rgba(0, 0, 0, 0.3); width:40%; margin:auto\">\n    The target\n</div>","metadata":{}},{"cell_type":"code","source":"if EDA:\n    df, hue  = train, None\n    if original is not None:\n        t, o = train.copy(), original.copy()\n        t['df'] = 'train'\n        o['df'] = 'original'\n        df, hue = pd.concat([t, o], axis=0), 'df'    \n    if PROBLEM == 'regression':\n        f = TARGET\n        fig, axs = plt.subplots(2, 2, figsize=(24, 10))\n        \n        ax = axs[0,0]\n        sns.kdeplot(train, x=f, label='train', ax=ax, color=red)\n        if original is not None:\n            sns.kdeplot(original[f], label='original', ax=ax, color=blue)\n        ax.set_title(f'{f} distribution')\n        ax.legend()\n        \n        ax = axs[0,1]\n        sns.boxplot(df, y=f, x=hue, ax=ax)\n        ax.set_title(f'{f} boxplot')\n        \n        ax = axs[1,0]\n        sns.kdeplot(np.log(train[f]), label='train', ax=ax, color=red)\n        if original is not None:\n            sns.kdeplot(np.log(original[f]), label='original', ax=ax, color=blue)\n        ax.set_title(f'log({f}) distribution')\n        ax.legend()\n        \n        ax = axs[1,1]\n        if original is not None:\n            sns.boxplot(y=np.log(df[f]), x=df['df'], ax=ax)\n        else :\n            sns.boxplot(y=np.log(df[f]), ax=ax)\n        ax.set_title(f'log({f}) boxplot')\n        \n        plt.tight_layout()\n        plt.show()\n    else:\n        fig, ax = plt.subplots(1, 1, figsize=(24, 5))\n        sns.countplot(train, x=TARGET)\n        ax.bar_label(ax.containers[0])\n        plt.tight_layout()\n        plt.show()\n","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:18:47.809431Z","iopub.execute_input":"2024-12-06T23:18:47.809753Z","iopub.status.idle":"2024-12-06T23:19:03.93956Z","shell.execute_reply.started":"2024-12-06T23:18:47.809718Z","shell.execute_reply":"2024-12-06T23:19:03.938462Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"display: block; padding: 3px 0; background-color: #ccffcc; color: black; font-size: 12px; font-weight: bold; border-radius: 12px; text-align: center; width: 100%; box-shadow: inset 4px 4px 8px rgba(255, 255, 255, 0.5), inset -4px -4px 8px rgba(0, 0, 0, 0.3); width:40%; margin:auto\">\n    The features\n</div>","metadata":{}},{"cell_type":"code","source":"def top_values(df : pd.DataFrame, column : str, n_max : int = 10, ax = None) :\n    vc = pd.DataFrame(df[column].value_counts()).reset_index()[:n_max]\n    sns.barplot(vc, y=column, x='count', ax=ax)\n    # titles\n    ax.set_title(f'{column} Top {n_max}');\n    ax.set_xlabel('')\n    ax.set_ylabel('')\n    ax.bar_label(ax.containers[0])\n\ndef counts_hist(df : pd.DataFrame, column : str, ax = None) :\n    vc = pd.DataFrame(df[column].value_counts()).reset_index()\n    sns.histplot(vc, x='count', ax=ax) # bins=int(vc.shape[0]/10), \n    # titles\n    ax.set_title(f'{column} counts histogram');\n    ax.set_xlabel('')\n    ax.set_ylabel('')\n\ndef corr(df, title='Correlation', ax=None):\n    associations_df = associations(df, nominal_columns='all', plot=False)\n    corr_matrix = associations_df['corr']\n    sns.heatmap(corr_matrix, annot=True, fmt='.2f', cmap='coolwarm', linewidths=0.5, ax=ax)\n    ax.set_title(title)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:19:03.941303Z","iopub.execute_input":"2024-12-06T23:19:03.942135Z","iopub.status.idle":"2024-12-06T23:19:03.953075Z","shell.execute_reply.started":"2024-12-06T23:19:03.942084Z","shell.execute_reply":"2024-12-06T23:19:03.951822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def top_target(df : pd.DataFrame, column : str, n_max : int = 10, ascending : bool = False, ax = None) :\n    vc = pd.DataFrame(df[column].value_counts()) #.reset_index()\n    tc = pd.DataFrame(df[column].loc[df[TARGET] == 1].value_counts())#.reset_index()\n    tc['ratio'] = tc['count'] * 100 / vc['count']\n    tc = tc.sort_values('ratio', ascending=ascending)\n    tc['ratio']  = np.round(tc['ratio'], 1)\n    if not ascending :\n        d = 'top'\n    else :\n        d = 'bottom'\n    tc = tc[:n_max]\n    sns.barplot(tc, y=tc.index, x='ratio', ax=ax)\n    # titles\n    ax.set_title(f'{column}, {d} {n_max} {TARGET} risk');\n    ax.set_xlabel('')\n    ax.set_ylabel(f'{TARGET} %')\n    ax.bar_label(ax.containers[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:19:03.95451Z","iopub.execute_input":"2024-12-06T23:19:03.95486Z","iopub.status.idle":"2024-12-06T23:19:03.976833Z","shell.execute_reply.started":"2024-12-06T23:19:03.954827Z","shell.execute_reply":"2024-12-06T23:19:03.975906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def top_target_w(df : pd.DataFrame, column : str, n_max : int = 10, ascending : bool = False, ax = None) :\n    vc = pd.DataFrame(df[column].value_counts()) #.reset_index()\n    tc = pd.DataFrame(df[column].loc[df[TARGET] == 1].value_counts())#.reset_index()\n    tc['ratio'] = tc['count'] * 100 * vc['count'] / df.shape[0]\n    tc = tc.sort_values('ratio', ascending=ascending)\n    tc['ratio']  = np.round(tc['ratio'], 1)\n    if not ascending :\n        d = 'top'\n    else :\n        d = 'bottom'\n    tc = tc[:n_max]\n    sns.barplot(tc, y=tc.index, x='ratio', ax=ax)\n    # titles\n    ax.set_title(f'{column}, {d} {n_max} {TARGET} risk');\n    ax.set_xlabel('')\n    ax.set_ylabel(f'{TARGET} % * Sample %')\n    ax.bar_label(ax.containers[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:19:03.97873Z","iopub.execute_input":"2024-12-06T23:19:03.979215Z","iopub.status.idle":"2024-12-06T23:19:03.992763Z","shell.execute_reply.started":"2024-12-06T23:19:03.979141Z","shell.execute_reply":"2024-12-06T23:19:03.991565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def kde_nums(train, test, original):\n    _, _, nums = fcn(test)\n    columns = nums\n    n_cols = 3\n    n_rows = math.ceil(len(columns)/n_cols)\n    fig, ax = plt.subplots(n_rows, n_cols, figsize=(16, n_rows*5))\n    ax = ax.flatten()\n\n    for i, column in enumerate(columns):\n        plot_axes = [ax[i]]\n\n        sns.kdeplot(train[column], label='Train', ax=ax[i], color=red)\n\n        sns.kdeplot(test[column], label='Test', ax=ax[i], color=blue)\n\n        if original is not None:\n            sns.kdeplot(original[column], label='Original', ax=ax[i], color=green)\n\n        # titles\n        ax[i].set_title(f'{column} Distribution');\n        ax[i].set_xlabel(None)\n\n    plt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:19:03.994667Z","iopub.execute_input":"2024-12-06T23:19:03.995055Z","iopub.status.idle":"2024-12-06T23:19:04.01024Z","shell.execute_reply.started":"2024-12-06T23:19:03.99501Z","shell.execute_reply":"2024-12-06T23:19:04.009042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if EDA:\n    print('Numerical features:')\n    kde_nums(train, test, original)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:19:04.011727Z","iopub.execute_input":"2024-12-06T23:19:04.012122Z","iopub.status.idle":"2024-12-06T23:20:30.596979Z","shell.execute_reply.started":"2024-12-06T23:19:04.012087Z","shell.execute_reply":"2024-12-06T23:20:30.595597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def count_cats(train_, test, n_cols=3, tight=True):\n\n    train = train_.copy()\n    train['nans'] = train['nans'].astype('string')\n    \n    columns = fcn(test)[1] + ['nans']\n    columns.remove('Policy Start Date')\n    if len(columns) == 0:\n        return\n        \n    n_cols = 2\n    n_rows = math.ceil(len(columns) * 2 / n_cols)\n    fig, ax = plt.subplots(n_rows, n_cols, figsize=(16, n_rows*5))\n    ax = ax.flatten()\n\n    for i, column in enumerate(columns):\n        n = i * n_cols\n        \n        vc = pd.DataFrame(train[column].value_counts()[:10]).reset_index()\n        sns.barplot(vc, x=column, y='count', ax=ax[n])\n\n        # titles\n        ax[n].set_title(column);\n        ax[n].set_xlabel('')\n        ax[n].set_ylabel('')\n        ax[n].bar_label(ax[n].containers[0])\n\n        sns.boxplot(train, y = TARGET, x = column, ax = ax[n + 1])\n\n    if tight:\n        plt.tight_layout()\n    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:20:30.598384Z","iopub.execute_input":"2024-12-06T23:20:30.598768Z","iopub.status.idle":"2024-12-06T23:20:30.609116Z","shell.execute_reply.started":"2024-12-06T23:20:30.598734Z","shell.execute_reply":"2024-12-06T23:20:30.60732Z"},"_kg_hide-input":false},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if EDA:\n    print('Categorical features')\n    count_cats(train, test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:20:30.611214Z","iopub.execute_input":"2024-12-06T23:20:30.611666Z","iopub.status.idle":"2024-12-06T23:20:45.87408Z","shell.execute_reply.started":"2024-12-06T23:20:30.611628Z","shell.execute_reply":"2024-12-06T23:20:45.872996Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"display: block; padding: 3px 0; background-color: #ccffcc; color: black; font-size: 12px; font-weight: bold; border-radius: 12px; text-align: center; width: 100%; box-shadow: inset 4px 4px 8px rgba(255, 255, 255, 0.5), inset -4px -4px 8px rgba(0, 0, 0, 0.3); width:40%; margin:auto\">\n    Correlation\n</div>","metadata":{}},{"cell_type":"code","source":"if EDA:\n    df_corr(train[:5000], name='TRAIN')     ","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:20:45.875625Z","iopub.execute_input":"2024-12-06T23:20:45.876109Z","iopub.status.idle":"2024-12-06T23:20:59.614904Z","shell.execute_reply.started":"2024-12-06T23:20:45.876046Z","shell.execute_reply":"2024-12-06T23:20:59.613674Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# &#128476; PREPROCESS","metadata":{}},{"cell_type":"markdown","source":"<div style=\"display: block; padding: 3px 0; background-color: #ccffcc; color: black; font-size: 12px; font-weight: bold; border-radius: 12px; text-align: center; width: 100%; box-shadow: inset 4px 4px 8px rgba(255, 255, 255, 0.5), inset -4px -4px 8px rgba(0, 0, 0, 0.3); width:40%; margin:auto\">\n    Set the base for dataset evaluation\n</div>","metadata":{}},{"cell_type":"code","source":"evaluate_dataset('loaded', update=False) ","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:20:59.616385Z","iopub.execute_input":"2024-12-06T23:20:59.616756Z","iopub.status.idle":"2024-12-06T23:21:03.362067Z","shell.execute_reply.started":"2024-12-06T23:20:59.616713Z","shell.execute_reply":"2024-12-06T23:21:03.36105Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"display: block; padding: 3px 0; background-color: #ccffcc; color: black; font-size: 12px; font-weight: bold; border-radius: 12px; text-align: center; width: 100%; box-shadow: inset 4px 4px 8px rgba(255, 255, 255, 0.5), inset -4px -4px 8px rgba(0, 0, 0, 0.3); width:40%; margin:auto\">\n    Add original\n</div>","metadata":{}},{"cell_type":"markdown","source":"* *and check if the CV score is better*\n* *optional, use original or not it depends*","metadata":{}},{"cell_type":"code","source":"def add_original(train_, original_):\n    f = train_.columns.tolist()\n    original_ = original_.dropna(subset = [TARGET])\n    train = pd.concat([train_, original_[f]]).reset_index(drop=True) # , axis=0, ignore_index=True\n    return train","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:21:03.363568Z","iopub.execute_input":"2024-12-06T23:21:03.364067Z","iopub.status.idle":"2024-12-06T23:21:03.370761Z","shell.execute_reply.started":"2024-12-06T23:21:03.364016Z","shell.execute_reply":"2024-12-06T23:21:03.369604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# if original is not None:\n#     train['original'] = 'no'\n#     test['original'] = 'no'\n#     original['original'] = 'yes'\n#     train_add = add_original(train, original)  \n#     DB.datasets.put('+original', train_add, test)\n#     DB.datasets.summary('+original')","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:21:03.372629Z","iopub.execute_input":"2024-12-06T23:21:03.373146Z","iopub.status.idle":"2024-12-06T23:21:03.383263Z","shell.execute_reply.started":"2024-12-06T23:21:03.373068Z","shell.execute_reply":"2024-12-06T23:21:03.382105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# if original is not None:    \n#     evaluate_dataset('+original')","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:21:03.384932Z","iopub.execute_input":"2024-12-06T23:21:03.385463Z","iopub.status.idle":"2024-12-06T23:21:03.397317Z","shell.execute_reply.started":"2024-12-06T23:21:03.385406Z","shell.execute_reply":"2024-12-06T23:21:03.39599Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## &#128205; Cleaning","metadata":{}},{"cell_type":"code","source":"def clean(src, dst):     \n    train, test = DB.datasets.train(src), DB.datasets.test(src)\n\n    for df in [train, test] :\n        df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n        df['Year'] = df['Policy Start Date'].dt.year.astype(float)    \n        df.drop('Policy Start Date', axis=1, inplace=True)    \n    DB.datasets.put(dst, train, test)\n\nclean('loaded', 'cleaned')\n\nDB.datasets.summary('cleaned')","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:21:03.399383Z","iopub.execute_input":"2024-12-06T23:21:03.399857Z","iopub.status.idle":"2024-12-06T23:21:13.906833Z","shell.execute_reply.started":"2024-12-06T23:21:03.39981Z","shell.execute_reply":"2024-12-06T23:21:13.905576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"evaluate_dataset('cleaned', update=False)","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:21:13.908277Z","iopub.execute_input":"2024-12-06T23:21:13.90863Z","iopub.status.idle":"2024-12-06T23:21:17.583881Z","shell.execute_reply.started":"2024-12-06T23:21:13.908595Z","shell.execute_reply":"2024-12-06T23:21:17.582608Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## &#128205; Feature engineering","metadata":{}},{"cell_type":"code","source":"src, dst = 'cleaned', 'fe'\ntrain, test = DB.datasets.train(src), DB.datasets.test(src)\n\nobject_columns = fcn(test)[1]\n\nfor df in [train, test] :\n    for column in object_columns :\n        df[column] = df[column].astype('string')\n        df['Year'] = df['Year'].astype(float)\n\nunknown = ['Marital Status', 'Occupation', 'Customer Feedback']\nfor df in [train, test] :\n    for column in  unknown:\n        df[column] = df[column].fillna('Unknown')\n\nn_cols, n_rows = len(unknown), 1\nfig, ax = plt.subplots(n_rows, n_cols, figsize=(16, n_rows*5))\nax = ax.flatten()\nfor i, column in enumerate(unknown):\n    sns.boxplot(train, y = TARGET, x = column, ax = ax[i])\nplt.show()\n\nsource = train.copy()\nsource['df'] = 'source'\nmedian = ['Age', 'Vehicle Age', 'Insurance Duration', 'Annual Income', 'Health Score', 'Previous Claims', 'Credit Score']\nfor df in [train, test] :\n    for column in  median:\n        mdn = train[column].median()\n        df[column] = df[column].fillna(mdn)\n        df['Number of Dependents'] = df['Number of Dependents'].fillna(2.0)\nsource = pd.concat([source, train])\nsource['df'] = source['df'].fillna('imputed')\n\nDB.datasets.put(dst, train, test)\nDB.datasets.summary(dst)","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:21:17.585242Z","iopub.execute_input":"2024-12-06T23:21:17.585606Z","iopub.status.idle":"2024-12-06T23:21:33.549472Z","shell.execute_reply.started":"2024-12-06T23:21:17.585561Z","shell.execute_reply":"2024-12-06T23:21:33.548286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"evaluate_dataset('fe', update=False) ","metadata":{"_kg_hide-input":false,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:21:33.55089Z","iopub.execute_input":"2024-12-06T23:21:33.551247Z","iopub.status.idle":"2024-12-06T23:21:37.149664Z","shell.execute_reply.started":"2024-12-06T23:21:33.551214Z","shell.execute_reply":"2024-12-06T23:21:37.148597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display_scores(DATASETS_SCORES)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:21:37.151093Z","iopub.execute_input":"2024-12-06T23:21:37.151574Z","iopub.status.idle":"2024-12-06T23:21:37.464627Z","shell.execute_reply.started":"2024-12-06T23:21:37.151526Z","shell.execute_reply":"2024-12-06T23:21:37.463433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    \nfrom sklearn.decomposition import PCA\n\ndef plot_variance(pca, width=10, height=4, dpi=100):\n    # Create figure\n    fig, axs = plt.subplots(1, 2)\n    n = pca.n_components_\n    grid = np.arange(1, n + 1)\n    # Explained variance\n    evr = pca.explained_variance_ratio_\n    axs[0].bar(grid, evr)\n    axs[0].set(\n        xlabel=\"Component\", title=\"% Explained Variance\", ylim=(0.0, 0.2)\n    )\n\n    # Cumulative Variance\n    cv = np.cumsum(evr)\n    axs[1].plot(np.r_[0, grid], np.r_[0, cv], \"o-\")\n    axs[1].set(\n        xlabel=\"Component\", title=\"\", ylim=(0.0, 1.0)\n    )\n    \n    # Set up figure\n    fig.set(figwidth=width, figheight=height,  dpi=100)\n    return axs\n\ndef make_pca(train, test, display_loadings=True, plots=True):\n    features = test.columns.tolist()\n    pca = PCA(2)\n    X_pca = pca.fit_transform(train.copy()[features])\n    X_test_pca = pca.transform(test.copy()[features])\n    component_names = [f\"PC{i+1}\" for i in range(X_pca.shape[1])]\n    X_pca = pd.DataFrame(X_pca, columns=component_names)\n    X_test_pca = pd.DataFrame(X_test_pca, columns=component_names)\n\n    if display_loadings:\n        loadings = pd.DataFrame(\n            pca.components_.T,  # transpose the matrix of loadings\n            columns=component_names,  # so the columns are the principal components\n            index=features,  # and the rows are the original features\n        )\n        print('\\nLoadings:')\n        display(loadings.head(20).style.background_gradient(subset=loadings.columns.to_list(), cmap='Greens'))\n\n    if plots:\n        plot_variance(pca)\n    return X_pca, X_test_pca","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T00:01:22.22549Z","iopub.execute_input":"2024-12-07T00:01:22.225932Z","iopub.status.idle":"2024-12-07T00:01:22.237348Z","shell.execute_reply.started":"2024-12-07T00:01:22.225897Z","shell.execute_reply":"2024-12-07T00:01:22.236078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if EDA:\n    train_pca, test_pca = make_pca(DB.datasets.train('fe_enccat'), DB.datasets.test('fe_enccat'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T00:01:30.644445Z","iopub.execute_input":"2024-12-07T00:01:30.644917Z","iopub.status.idle":"2024-12-07T00:01:31.953634Z","shell.execute_reply.started":"2024-12-07T00:01:30.644878Z","shell.execute_reply":"2024-12-07T00:01:31.952534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef mutual_info(train_df, plot=True):\n    train = train_df.copy()\n    X_train = train.drop([TARGET], axis=1) # [features]\n    y_train = train[TARGET]\n\n    if PROBLEM == 'regression':\n        mi_scores = mutual_info_regression(X_train, y_train, random_state=42)\n    else:\n        mi_scores = mutual_info_classif(X_train, y_train, random_state=42)\n\n    mi_scores = pd.Series(mi_scores, name=\"MI_score\", index=X_train.columns)\n    mi_scores = mi_scores.sort_values(ascending=False)\n    df_mi_scores = pd.DataFrame(mi_scores).reset_index().rename(columns={'index':'feature'})\n    display(df_mi_scores.style.background_gradient(subset=['MI_score'], cmap='Reds'))\n    plt.figure(figsize=(24, 16))\n    d = sns.barplot(y=df_mi_scores['feature'], x=df_mi_scores['MI_score'])\n    return mi_scores\n\nif EDA:\n    mi_scores = mutual_info(DB.datasets.train('fe_enccat').sample(10000))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T23:37:55.56122Z","iopub.execute_input":"2024-12-06T23:37:55.561767Z","iopub.status.idle":"2024-12-06T23:37:57.936782Z","shell.execute_reply.started":"2024-12-06T23:37:55.561724Z","shell.execute_reply":"2024-12-06T23:37:57.935538Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# &#128204; CV ","metadata":{}},{"cell_type":"markdown","source":"## &#128204; *Datasets*","metadata":{}},{"cell_type":"code","source":"# tocat('cleaned', 'cleaned_cat')\n# allcat('cleaned', 'cleaned_allcat')\n# enccat('cleaned', 'cleaned_enccat')\n# DB.save()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:14.69533Z","iopub.execute_input":"2024-12-06T15:26:14.695591Z","iopub.status.idle":"2024-12-06T15:26:14.699453Z","shell.execute_reply.started":"2024-12-06T15:26:14.695564Z","shell.execute_reply":"2024-12-06T15:26:14.698556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# enccat('fe', 'fe_enccat')\n# tocat('fe', 'fe_cat')\n# DB.save()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:14.700536Z","iopub.execute_input":"2024-12-06T15:26:14.700827Z","iopub.status.idle":"2024-12-06T15:26:14.712315Z","shell.execute_reply.started":"2024-12-06T15:26:14.700802Z","shell.execute_reply":"2024-12-06T15:26:14.711421Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## &#128204; *Saved CVs*","metadata":{}},{"cell_type":"code","source":"scores = {}\nTRAIN, TEST, TRAINP, TESTP = pd.DataFrame(), pd.DataFrame(), pd.DataFrame(), pd.DataFrame()\nTRAIN[TARGET] = DB.datasets.y('cleaned')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:14.713438Z","iopub.execute_input":"2024-12-06T15:26:14.713732Z","iopub.status.idle":"2024-12-06T15:26:14.831598Z","shell.execute_reply.started":"2024-12-06T15:26:14.713701Z","shell.execute_reply":"2024-12-06T15:26:14.830635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RERUN = False\n\nfor _, cv in DB.cvs.data_.items() :\n\n    summ = cv.summary\n    estimator_class = summ['estimator_class']\n    name = summ['name']\n\n    if name.startswith('dataset_eval') :\n        continue\n\n    display(summ)\n\n    params = summ['params']\n    \n\n    if not RERUN :\n        print('\\n========== model :', estimator_class, '==========\\n')\n        estimator = ESTIMATORS[estimator_class](params=params)\n        estimator.upload_cv(cv)\n        if SHOW_CVs :    \n            cv.display(estimator)\n        cv_scores = estimator.cv_scores()\n        cv_scores['Dataset'] = summ['dataset_name']\n        scores[name] = cv_scores\n        \n        #  oof data for stack and blend\n        TRAIN[name], TEST[name] = cv.oof['train'], cv.oof['test']\n    else:\n        estimator = ESTIMATORS[estimator_class](params=params)\n        if type(params) is not dict :\n            continue\n\n        print(colored(f\"\\n{estimator_class}:\", 'blue'))\n        for  k, v in params.items() :\n            print('    ', k, ' '*(30-len(k)), ':', v)\n\n        new_cv = cv.run(estimator, DB)\n        DB.cvs.put(new_cv)\n        \n        TRAIN[name]  = new_cv.oof['train']\n        TEST[name] = new_cv.oof['test']        \n        \n        scores[name] = estimator.cv_scores()\n        dbnew.save\n\n    # estimator.submit()\n    \ndisplay_scores(scores)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:14.833162Z","iopub.execute_input":"2024-12-06T15:26:14.833508Z","iopub.status.idle":"2024-12-06T15:26:44.722949Z","shell.execute_reply.started":"2024-12-06T15:26:14.833471Z","shell.execute_reply":"2024-12-06T15:26:44.722093Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## &#128204; *New CV*","metadata":{}},{"cell_type":"code","source":"SPLITS, REPEATS = 5, 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:44.724272Z","iopub.execute_input":"2024-12-06T15:26:44.724645Z","iopub.status.idle":"2024-12-06T15:26:44.729179Z","shell.execute_reply.started":"2024-12-06T15:26:44.724606Z","shell.execute_reply":"2024-12-06T15:26:44.728245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# name, dataset = 'CAT_0', 'cleaned_allcat'\n\n# params = {\n#     'iterations': 1000,\n#     # 'learning_rate' : 0.05,\n#     'random_seed':42,\n#     'use_best_model'    : True,\n#     'od_type'           : 'Iter',\n#     'od_wait'           : 20,\n#     'cat_features'  : fcn(DB.datasets.test(dataset))[1],\n# }\n\n# estimator = CATRegressor(params=params)\n# cv = estimator.crossvalidate(\n#         DB.datasets.train(dataset),\n#         DB.datasets.test(dataset),\n#         dataset_name = dataset, \n#         n_splits = SPLITS,\n#         n_repeats = REPEATS,\n#         clear = False,\n#         verbose = 2,\n#         name = name,\n# )\n# scores[name] = estimator.cv_scores()\n\n# # compare with saved\n# display_scores(scores)\n\n# TRAIN[name]  = cv.oof['train']\n# TEST[name] = cv.oof['test']\n\n# DB.cvs.put(cv)\n# # DB.save()\n\n# # estimator.submit()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:44.730327Z","iopub.execute_input":"2024-12-06T15:26:44.730684Z","iopub.status.idle":"2024-12-06T15:26:44.742989Z","shell.execute_reply.started":"2024-12-06T15:26:44.730627Z","shell.execute_reply":"2024-12-06T15:26:44.74216Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# name, dataset = 'CAT_1', 'fe_cat'\n\n# params = {\n#     'loss_function'     : 'RMSE',\n#     'iterations'        : 300,\n#     # 'learning_rate'     : 0.05,\n#     'random_seed'       : 42,\n#     'use_best_model'    : True,\n#     'od_type'           : 'Iter',\n#     'od_wait'           : 20,\n#     'cat_features'  : fcn(DB.datasets.test(dataset))[1],\n# }\n\n# estimator = CATRegressor(params=params)\n# cv = estimator.crossvalidate(\n#         DB.datasets.train(dataset),\n#         DB.datasets.test(dataset),\n#         dataset_name = dataset, \n#         n_splits = SPLITS,\n#         n_repeats = REPEATS,\n#         clear = False,\n#         verbose = 2,\n#         name = name,\n# )\n# scores[name] = estimator.cv_scores()\n\n# # compare with saved\n# display_scores(scores)\n\n# TRAIN[name]  = cv.oof['train']\n# TEST[name] = cv.oof['test']\n\n# DB.cvs.put(cv)\n# # DB.save()\n\n# # estimator.submit()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:44.743959Z","iopub.execute_input":"2024-12-06T15:26:44.74426Z","iopub.status.idle":"2024-12-06T15:26:44.756295Z","shell.execute_reply.started":"2024-12-06T15:26:44.744234Z","shell.execute_reply":"2024-12-06T15:26:44.755582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef addbest(src, dst):     \n    train, test = DB.datasets.train(src), DB.datasets.test(src)\n    train_b, test_b = DB.datasets.train('best'), DB.datasets.test('best')\n    train['best'] = train_b['best']\n    test['best'] = test_b['best']\n    DB.datasets.put(dst, train, test)","metadata":{"_kg_hide-input":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:44.757209Z","iopub.execute_input":"2024-12-06T15:26:44.757483Z","iopub.status.idle":"2024-12-06T15:26:44.767797Z","shell.execute_reply.started":"2024-12-06T15:26:44.757458Z","shell.execute_reply":"2024-12-06T15:26:44.767075Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# addbest('fe_cat', 'fe_cat_best')\n# addbest('fe_enccat', 'fe_enccat_best')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:44.768775Z","iopub.execute_input":"2024-12-06T15:26:44.769356Z","iopub.status.idle":"2024-12-06T15:26:44.777346Z","shell.execute_reply.started":"2024-12-06T15:26:44.769309Z","shell.execute_reply":"2024-12-06T15:26:44.77671Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# name, dataset = 'CAT_2', 'fe_cat_best'\n\n# params = {\n#     'loss_function'     : 'RMSE',\n#     'iterations'        : 300,\n#     # 'learning_rate'     : 0.05,\n#     'random_seed'       : 42,\n#     'use_best_model'    : True,\n#     'od_type'           : 'Iter',\n#     'od_wait'           : 20,\n#     'cat_features'  : fcn(DB.datasets.test(dataset))[1],\n# }\n\n# estimator = CATRegressor(params=params)\n# cv = estimator.crossvalidate(\n#         DB.datasets.train(dataset),\n#         DB.datasets.test(dataset),\n#         dataset_name = dataset, \n#         n_splits = SPLITS,\n#         n_repeats = REPEATS,\n#         clear = False,\n#         verbose = 2,\n#         name = name,\n# )\n# scores[name] = estimator.cv_scores()\n\n# # compare with saved\n# display_scores(scores)\n\n# TRAIN[name]  = cv.oof['train']\n# TEST[name] = cv.oof['test']\n\n# DB.cvs.put(cv)\n# # DB.save()\n\n# # estimator.submit()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:44.778374Z","iopub.execute_input":"2024-12-06T15:26:44.779132Z","iopub.status.idle":"2024-12-06T15:26:44.791351Z","shell.execute_reply.started":"2024-12-06T15:26:44.779105Z","shell.execute_reply":"2024-12-06T15:26:44.790621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# name, dataset = 'CAT_3', 'fe_cat_best'\n\n# params = {\n#     'loss_function'     : 'RMSE',\n#     'iterations'        : 3000,\n#     # 'learning_rate'     : 0.05,\n#     'random_seed'       : 42,\n#     'use_best_model'    : True,\n#     'od_type'           : 'Iter',\n#     'od_wait'           : 20,\n#     'cat_features'  : fcn(DB.datasets.test(dataset))[1],\n# }\n\n# estimator = CATRegressor(params=params)\n# cv = estimator.crossvalidate(\n#         DB.datasets.train(dataset),\n#         DB.datasets.test(dataset),\n#         dataset_name = dataset, \n#         n_splits = SPLITS,\n#         n_repeats = REPEATS,\n#         clear = False,\n#         verbose = 2,\n#         name = name,\n# )\n# scores[name] = estimator.cv_scores()\n\n# # compare with saved\n# display_scores(scores)\n\n# TRAIN[name]  = cv.oof['train']\n# TEST[name] = cv.oof['test']\n\n# DB.cvs.put(cv)\n# # DB.save()\n\n# estimator.submit()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:44.792158Z","iopub.execute_input":"2024-12-06T15:26:44.792393Z","iopub.status.idle":"2024-12-06T15:26:44.802597Z","shell.execute_reply.started":"2024-12-06T15:26:44.792369Z","shell.execute_reply":"2024-12-06T15:26:44.801941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# name, dataset = 'LGB_0', 'cleaned_enccat'\n\n# params = {\n#     # 'n_estimators': 1000,\n#     'learning_rate' : 0.05,\n#     'random_seed':42,\n# }\n\n# estimator = LGBRegressor(params=params)\n# cv = estimator.crossvalidate(\n#         DB.datasets.train(dataset),\n#         DB.datasets.test(dataset),\n#         dataset_name = dataset, \n#         n_splits = SPLITS,\n#         n_repeats = REPEATS,\n#         clear = True,\n#         verbose = 2,\n#         name = name,\n# )\n# scores[name] = estimator.cv_scores()\n\n# # compare with saved\n# display_scores(scores)\n\n# TRAIN[name]  = cv.oof['train']\n# TEST[name] = cv.oof['test']\n\n# DB.cvs.put(cv)\n# # DB.save()\n\n# # estimator.submit()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:44.803574Z","iopub.execute_input":"2024-12-06T15:26:44.803829Z","iopub.status.idle":"2024-12-06T15:26:44.8121Z","shell.execute_reply.started":"2024-12-06T15:26:44.803806Z","shell.execute_reply":"2024-12-06T15:26:44.811386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# name, dataset = 'LGB_1', 'fe_enccat'\n\n# params = {\n#     'n_esimators': 3000,\n#     'learning_rate' : 0.05,\n#     'random_seed':42,\n# }\n\n# estimator = LGBRegressor(params=params)\n# cv = estimator.crossvalidate(\n#         DB.datasets.train(dataset),\n#         DB.datasets.test(dataset),\n#         dataset_name = dataset, \n#         n_splits = SPLITS,\n#         n_repeats = REPEATS,\n#         clear = True,\n#         verbose = 2,\n#         name = name,\n# )\n# scores[name] = estimator.cv_scores()\n\n# # compare with saved\n# display_scores(scores)\n\n# TRAIN[name]  = cv.oof['train']\n# TEST[name] = cv.oof['test']\n\n# DB.cvs.put(cv)\n# # DB.save()\n\n# # estimator.submit()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:44.813006Z","iopub.execute_input":"2024-12-06T15:26:44.813338Z","iopub.status.idle":"2024-12-06T15:26:44.826991Z","shell.execute_reply.started":"2024-12-06T15:26:44.813292Z","shell.execute_reply":"2024-12-06T15:26:44.826223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# name, dataset = 'LGB_2', 'fe_enccat_best'\n\n# params = {\n#     'n_estimators': 3000,\n#     'learning_rate' : 0.05,\n#     'random_seed':42,\n# }\n\n# estimator = LGBRegressor(params=params)\n# cv = estimator.crossvalidate(\n#         DB.datasets.train(dataset),\n#         DB.datasets.test(dataset),\n#         dataset_name = dataset, \n#         n_splits = SPLITS,\n#         n_repeats = REPEATS,\n#         clear = True,\n#         verbose = 2,\n#         name = name,\n# )\n# scores[name] = estimator.cv_scores()\n\n# # compare with saved\n# display_scores(scores)\n\n# TRAIN[name]  = cv.oof['train']\n# TEST[name] = cv.oof['test']\n\n# DB.cvs.put(cv)\n# # DB.save()\n\n# # estimator.submit()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:44.827895Z","iopub.execute_input":"2024-12-06T15:26:44.828154Z","iopub.status.idle":"2024-12-06T15:26:44.837314Z","shell.execute_reply.started":"2024-12-06T15:26:44.828126Z","shell.execute_reply":"2024-12-06T15:26:44.836711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# name, dataset = 'XGB_0', 'cleaned_enccat'\n\n# params = {\n#     # 'n_esimators': 1000,\n#     'learning_rate' : 0.05,\n#     'random_seed':42,\n# }\n\n# estimator = xGBRegressor(params=params)\n# cv = estimator.crossvalidate(\n#         DB.datasets.train(dataset),\n#         DB.datasets.test(dataset),\n#         dataset_name = dataset, \n#         n_splits = SPLITS,\n#         n_repeats = REPEATS,\n#         clear = True,\n#         verbose = 2,\n#         name = name,\n# )\n# scores[name] = estimator.cv_scores()\n\n# # compare with saved\n# display_scores(scores)\n\n# TRAIN[name]  = cv.oof['train']\n# TEST[name] = cv.oof['test']\n\n# DB.cvs.put(cv)\n# # DB.save()\n\n# # estimator.submit()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:44.838224Z","iopub.execute_input":"2024-12-06T15:26:44.838483Z","iopub.status.idle":"2024-12-06T15:26:44.846272Z","shell.execute_reply.started":"2024-12-06T15:26:44.838453Z","shell.execute_reply":"2024-12-06T15:26:44.84562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# name, dataset = 'XGB_1', 'fe_enccat'\n\n# params = {\n#     'n_estimators': 5000,\n#     'learning_rate' : 0.05,\n#     'random_seed':42,\n# }\n\n# estimator = xGBRegressor(params=params)\n# cv = estimator.crossvalidate(\n#         DB.datasets.train(dataset),\n#         DB.datasets.test(dataset),\n#         dataset_name = dataset, \n#         n_splits = SPLITS,\n#         n_repeats = REPEATS,\n#         clear = True,\n#         verbose = 2,\n#         name = name,\n# )\n# scores[name] = estimator.cv_scores()\n\n# # compare with saved\n# display_scores(scores)\n\n# TRAIN[name]  = cv.oof['train']\n# TEST[name] = cv.oof['test']\n\n# DB.cvs.put(cv)\n# # DB.save()\n\n# # estimator.submit()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:44.847071Z","iopub.execute_input":"2024-12-06T15:26:44.847313Z","iopub.status.idle":"2024-12-06T15:26:44.860115Z","shell.execute_reply.started":"2024-12-06T15:26:44.847288Z","shell.execute_reply":"2024-12-06T15:26:44.859289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# name, dataset = 'XGB_2', 'fe_enccat_best'\n\n# params = {\n#     'n_estimators': 5000,\n#     'learning_rate' : 0.05,\n#     'random_seed':42,\n# }\n\n# estimator = xGBRegressor(params=params)\n# cv = estimator.crossvalidate(\n#         DB.datasets.train(dataset),\n#         DB.datasets.test(dataset),\n#         dataset_name = dataset, \n#         n_splits = SPLITS,\n#         n_repeats = REPEATS,\n#         clear = True,\n#         verbose = 2,\n#         name = name,\n# )\n# scores[name] = estimator.cv_scores()\n\n# # compare with saved\n# display_scores(scores)\n\n# TRAIN[name]  = cv.oof['train']\n# TEST[name] = cv.oof['test']\n\n# DB.cvs.put(cv)\n# DB.save()\n\n# # estimator.submit()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:44.861041Z","iopub.execute_input":"2024-12-06T15:26:44.861286Z","iopub.status.idle":"2024-12-06T15:26:44.871943Z","shell.execute_reply.started":"2024-12-06T15:26:44.861262Z","shell.execute_reply":"2024-12-06T15:26:44.871139Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## &#128204; *OOF Data*","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 12))\nsns.heatmap(TRAIN.corr(), annot=True, fmt='.2f')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-12-06T15:26:44.872985Z","iopub.execute_input":"2024-12-06T15:26:44.873875Z","iopub.status.idle":"2024-12-06T15:26:45.860747Z","shell.execute_reply.started":"2024-12-06T15:26:44.873848Z","shell.execute_reply":"2024-12-06T15:26:45.859884Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.DataFrame()\nfor c in TEST.columns.tolist() :\n    x = pd.DataFrame()\n    x['oof_proba'] = TRAIN[c]\n    x['model'] = c\n    df = pd.concat([df, x], axis=0)\n                   \nplt.figure(figsize=(16, 5))\nsns.kdeplot(TEST, bw_adjust=0.1, log_scale=(False, False))\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:26:45.861722Z","iopub.execute_input":"2024-12-06T15:26:45.861975Z","iopub.status.idle":"2024-12-06T15:27:22.184403Z","shell.execute_reply.started":"2024-12-06T15:26:45.86195Z","shell.execute_reply":"2024-12-06T15:27:22.18373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = TRAIN.copy()\nX_train = df.drop([TARGET], axis=1) # [features]\ny_train = df[TARGET]\n\nif PROBLEM == 'regression':\n    mi_scores = mutual_info_regression(X_train, y_train, random_state=42)\nelse:\n    mi_scores = mutual_info_classif(X_train, y_train, random_state=42)\n\nmi_scores = pd.Series(mi_scores, name=\"MI_score\", index=X_train.columns)\nmi_scores = mi_scores.sort_values(ascending=False)\ndf_mi_scores = pd.DataFrame(mi_scores).reset_index().rename(columns={'index':'feature'})\ndisplay(df_mi_scores.style.background_gradient(subset=['MI_score'], cmap='Reds'))\nplt.figure(figsize=(24, 16))\nd = sns.barplot(y=df_mi_scores['feature'], x=df_mi_scores['MI_score'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:27:22.185885Z","iopub.execute_input":"2024-12-06T15:27:22.18624Z","iopub.status.idle":"2024-12-06T15:29:19.530799Z","shell.execute_reply.started":"2024-12-06T15:27:22.186201Z","shell.execute_reply":"2024-12-06T15:29:19.529944Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## &#128204; STACK\n#### members list otimization","metadata":{}},{"cell_type":"code","source":"import itertools\ndef best_members(members, f_, constant_members=[], n_min=2, n_max=0, verbose=1, name='estimator'):\n    \n    def f(x) :\n        return OBJ * f_(x)\n    \n    best, f_best = [], -np.inf\n    \n    all, history = [], []\n    stop = len(members) + 1\n    if n_max > 0:\n        stop = n_max\n    for i in range(n_min, stop):\n        all_i = list(itertools.combinations(members, i))\n        all.extend(all_i)\n    trials = len(all)\n        \n    trial = 1\n    for curr in all:\n        is_best = False\n        curr = list(curr)\n        curr.extend(constant_members)\n        f_curr = f(curr)\n        if verbose > 0:\n            print(f'{trial}/{trials}', curr, ' '*(50-len(str(curr))), '->', OBJ * f_curr, end=' ')\n        if f_curr > f_best:\n            f_best = f_curr\n            best = curr\n            is_best = True\n            if verbose > 0:\n                print('best so far', end=' ')\n        if verbose > 0:\n            print()\n        history.append({\n            'trial'    : trial,\n            'members'  : str(curr),\n            'best'     : is_best,\n            SCORE_NAME : OBJ * f_curr,\n            'estimator': name,\n        })\n        trial += 1\n    if verbose > 0:\n        print('\\n\\nthe best:\\n', best, '->', OBJ * f_best)\n    return best, f_best, history","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:29:19.53208Z","iopub.execute_input":"2024-12-06T15:29:19.532442Z","iopub.status.idle":"2024-12-06T15:29:19.541586Z","shell.execute_reply.started":"2024-12-06T15:29:19.532405Z","shell.execute_reply":"2024-12-06T15:29:19.540709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def stack(train, test, members, verbose=2, name='STACK', submission=False):\n    \n    # estimator = RegressorWrapper(LinearRegression(), name=name)\n    # estimator = xGBRegressor(params={\n    estimator = LGBRegressor(params={\n        'n_estimators': 10000,\n        # 'learning_rate' : 0.01,\n        'random_seed':42,    \n    })\n    cv = estimator.crossvalidate(\n        train[members + [TARGET]], \n        test[members], \n        dataset_name=str(members), \n        n_splits=5,\n        n_repeats=1,\n        verbose=verbose,\n        clear=False,\n        name=name,\n    )\n    if submission:\n        estimator.submit()\n    return estimator.cv_scores(), cv.oof['train'], cv.oof['test']\n\ndef stack_score(members):\n    scores, _, _ = stack(TRAIN, TEST, members, verbose=0)\n    return scores[SCORE_NAME]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:29:19.542853Z","iopub.execute_input":"2024-12-06T15:29:19.543743Z","iopub.status.idle":"2024-12-06T15:29:19.556725Z","shell.execute_reply.started":"2024-12-06T15:29:19.543703Z","shell.execute_reply":"2024-12-06T15:29:19.555876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"members = TEST.columns.tolist() \nmembers","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:29:19.557905Z","iopub.execute_input":"2024-12-06T15:29:19.558228Z","iopub.status.idle":"2024-12-06T15:29:19.572061Z","shell.execute_reply.started":"2024-12-06T15:29:19.558193Z","shell.execute_reply":"2024-12-06T15:29:19.571316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# best, best_score, stack_history  = best_members(members, stack_score, name = 'linear regression')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:29:19.573279Z","iopub.execute_input":"2024-12-06T15:29:19.573862Z","iopub.status.idle":"2024-12-06T15:29:19.580083Z","shell.execute_reply.started":"2024-12-06T15:29:19.573823Z","shell.execute_reply":"2024-12-06T15:29:19.579442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score, train_best, test_best = stack(TRAIN, TEST, members, submission=True)\nscores['STACK'] = score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:29:19.58104Z","iopub.execute_input":"2024-12-06T15:29:19.581364Z","iopub.status.idle":"2024-12-06T15:29:58.452439Z","shell.execute_reply.started":"2024-12-06T15:29:19.581329Z","shell.execute_reply":"2024-12-06T15:29:58.451709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# trn, tst = pd.DataFrame(), pd.DataFrame()\n# trn['best'] = train_best\n# tst['best'] = test_best\n# DB.datasets.put('best', trn, tst)\n# DB.save()\n# DB","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:29:58.453558Z","iopub.execute_input":"2024-12-06T15:29:58.453848Z","iopub.status.idle":"2024-12-06T15:29:58.457828Z","shell.execute_reply.started":"2024-12-06T15:29:58.453821Z","shell.execute_reply":"2024-12-06T15:29:58.456875Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## &#128204; BLEND *differential evolution*\n#### members list and weights otimization","metadata":{}},{"cell_type":"code","source":"def blend_diff(train, test, members, verbose=1):\n    \n    def weighted_average(weights, values):\n        qty = len(values)\n        sum_values = values[0] * weights[0]\n        sum_weights = weights[0]\n        for i in range(1, qty):\n            sum_values += values[i] * weights[i]\n            sum_weights += weights[i]\n        return sum_values / sum_weights\n\n    def obj(weights):\n        preds = weighted_average(weights, X)\n        return LOSS(y, preds)    \n \n    X = [train[col].values for col in members]\n    if verbose > 0:\n        print('train arrays:')\n        for a in X:\n            print(a)\n            \n    maxiter = len(members) * 1000\n    \n    y = train[TARGET].values\n    \n    qty = len(X)\n    initial_weights = [1 for _ in range(qty)]\n    bounds = [(0.0, 1.0) for _ in range(qty)]\n    result = sp.optimize.differential_evolution(obj, bounds=bounds, maxiter=1000*qty)\n    weights = result.x\n    \n    if verbose > 0 :\n        print('\\n', obj(initial_weights), end=' => ')\n        print(weights)\n        print() \n        \n    score = SCORE(y, weighted_average(weights, X))\n    if verbose > 0:\n        print(colored(f'estimated {SCORE_NAME} : {score}\\n\\n', 'red'))\n\n    X_test = [test[col].values for col in members]\n    if verbose > 0:\n        print('arrays to blend:')\n        for a in X_test:\n            print(a)\n\n    result = weighted_average(weights, X_test)\n    if verbose > 0:\n        print('weighted average:\\n', result, '\\n')\n    \n    return result, score\n\ndef blend_score_diff(members):\n    _, score = blend_diff(TRAIN, TEST, members, verbose=0)\n    return score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:29:58.458726Z","iopub.execute_input":"2024-12-06T15:29:58.459Z","iopub.status.idle":"2024-12-06T15:29:58.470567Z","shell.execute_reply.started":"2024-12-06T15:29:58.458974Z","shell.execute_reply":"2024-12-06T15:29:58.469712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"members[-3:]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:29:58.471585Z","iopub.execute_input":"2024-12-06T15:29:58.47186Z","iopub.status.idle":"2024-12-06T15:29:58.484797Z","shell.execute_reply.started":"2024-12-06T15:29:58.471836Z","shell.execute_reply":"2024-12-06T15:29:58.483892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# best, best_score, diff_history  = best_members(members, blend_score_diff, name='differential evolution')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:29:58.48576Z","iopub.execute_input":"2024-12-06T15:29:58.486023Z","iopub.status.idle":"2024-12-06T15:29:58.496019Z","shell.execute_reply.started":"2024-12-06T15:29:58.485991Z","shell.execute_reply":"2024-12-06T15:29:58.495272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TEST_PREDS_D, score_diff = blend_diff(TRAIN, TEST, members) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:29:58.496846Z","iopub.execute_input":"2024-12-06T15:29:58.497055Z","iopub.status.idle":"2024-12-06T15:30:17.006642Z","shell.execute_reply.started":"2024-12-06T15:29:58.497033Z","shell.execute_reply":"2024-12-06T15:30:17.005807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores['DIFF_EV'] = {\n    'Model' : 'DIFF_EV',\n    SCORE_NAME : score_diff,\n}\n\n# TEST['DIFF'] = TEST_PROBAS_D","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:30:17.007729Z","iopub.execute_input":"2024-12-06T15:30:17.008Z","iopub.status.idle":"2024-12-06T15:30:17.01226Z","shell.execute_reply.started":"2024-12-06T15:30:17.007973Z","shell.execute_reply":"2024-12-06T15:30:17.01124Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = sample_sub.copy()\nsub[TARGET] = TEST_PREDS_D\nsub.to_csv(f\"diff_ev_{score_diff}.csv\", index=False)\ndisplay(sub.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:30:17.013543Z","iopub.execute_input":"2024-12-06T15:30:17.013944Z","iopub.status.idle":"2024-12-06T15:30:18.428307Z","shell.execute_reply.started":"2024-12-06T15:30:17.013906Z","shell.execute_reply":"2024-12-06T15:30:18.426876Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## &#128204; BLEND *sp.optimize.minimize*\n#### members list and weights otimization","metadata":{}},{"cell_type":"code","source":"def blend(train, test, members, verbose=1):\n    X = [train[col].values for col in members]\n    if verbose > 0:\n        print('train arrays:')\n        for a in X:\n            print(a)\n            \n    maxiter = len(members) * 500\n    \n    y = train[TARGET].values\n    \n    averager = Averager(options={'maxiter':maxiter})\n    \n    x_valid = np.clip(averager.fit_predict(X, y), 20.0, 4999.0)\n    \n    if verbose > 0:\n        print('\\nWEIGHTS:\\n', averager.weights())\n        print('weighted average:\\n', x_valid)\n\n    score = SCORE(y, x_valid)\n    if verbose > 0:\n        print(colored(f'estimated {SCORE_NAME} : {score}\\n\\n', 'red'))\n\n    X_test = [test[col].values for col in members]\n    if verbose > 0:\n        print('arrays to blend:')\n        for a in X_test:\n            print(a)\n\n    result = np.clip(averager.predict(X_test), 20, 4999)\n    if verbose > 0:\n        print('weighted average:\\n', result, '\\n')\n    \n    return result, score\n\ndef blend_score(members):\n    _, score = blend(TRAIN, TEST, members, verbose=0)\n    return score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:30:18.439596Z","iopub.execute_input":"2024-12-06T15:30:18.439944Z","iopub.status.idle":"2024-12-06T15:30:18.448167Z","shell.execute_reply.started":"2024-12-06T15:30:18.439915Z","shell.execute_reply":"2024-12-06T15:30:18.447165Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# best, best_score, blend_history = best_members(members, blend_score, name='sp.optimize.minimize')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:30:18.44921Z","iopub.execute_input":"2024-12-06T15:30:18.449535Z","iopub.status.idle":"2024-12-06T15:30:18.461421Z","shell.execute_reply.started":"2024-12-06T15:30:18.449493Z","shell.execute_reply":"2024-12-06T15:30:18.460578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TEST_PREDS_A, score_averager = blend(TRAIN, TEST, members) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:30:18.46236Z","iopub.execute_input":"2024-12-06T15:30:18.462602Z","iopub.status.idle":"2024-12-06T15:31:48.706348Z","shell.execute_reply.started":"2024-12-06T15:30:18.462578Z","shell.execute_reply":"2024-12-06T15:31:48.7054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scores['BLEND'] = {\n    'Model' : 'BLEND',\n    SCORE_NAME : score_averager,\n}\ndisplay_scores(scores)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:31:48.707416Z","iopub.execute_input":"2024-12-06T15:31:48.707693Z","iopub.status.idle":"2024-12-06T15:31:49.178921Z","shell.execute_reply.started":"2024-12-06T15:31:48.707667Z","shell.execute_reply":"2024-12-06T15:31:49.178071Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## &#128204; Compare ensembles","metadata":{}},{"cell_type":"code","source":"\n# dfs = pd.DataFrame(stack_history)\n# dfb = pd.DataFrame(blend_history)\n# dfd = pd.DataFrame(diff_history)\n# df = pd.concat([dfs, dfb, dfd], axis=0)\n\n# mn, mx = df[SCORE_NAME].min(), df[SCORE_NAME].max()\n# fig, axs = plt.subplots(2, 1, figsize=(16, 18))\n# axs = axs.flatten()\n\n# ax = axs[0]\n# sns.barplot(df, y='members', x=SCORE_NAME, hue='estimator', palette=[green, red, blue, black], ax=ax)\n# ax.set(xlim=(mn*0.9999, mx*1.0001))\n# ax.legend = False\n\n# ax = axs[1]\n# sns.lineplot(df, x='trial', y=SCORE_NAME, hue='estimator', palette=[green, red, blue, black], ax=ax)\n# ax.set(ylim=(mn*0.9999, mx*1.0001))\n# plt.tight_layout()\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:31:49.180108Z","iopub.execute_input":"2024-12-06T15:31:49.180467Z","iopub.status.idle":"2024-12-06T15:31:49.185232Z","shell.execute_reply.started":"2024-12-06T15:31:49.180427Z","shell.execute_reply":"2024-12-06T15:31:49.18427Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### &#128190; SUBMISSION","metadata":{}},{"cell_type":"code","source":"sub = sample_sub.copy()\nsub[TARGET] = TEST_PREDS_A\nsub.to_csv(f\"averager_{score_averager}.csv\", index=False)\ndisplay(sub.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:31:49.186262Z","iopub.execute_input":"2024-12-06T15:31:49.186507Z","iopub.status.idle":"2024-12-06T15:31:50.545451Z","shell.execute_reply.started":"2024-12-06T15:31:49.186482Z","shell.execute_reply":"2024-12-06T15:31:50.544558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DB.save()\nDB","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T15:31:50.546455Z","iopub.execute_input":"2024-12-06T15:31:50.546758Z","iopub.status.idle":"2024-12-06T15:40:25.749104Z","shell.execute_reply.started":"2024-12-06T15:31:50.546731Z","shell.execute_reply":"2024-12-06T15:40:25.748222Z"}},"outputs":[],"execution_count":null}]}